Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions docs/PROVIDERS.md
Original file line number Diff line number Diff line change
Expand Up @@ -87,6 +87,8 @@ Custom pricing overlays are exact-match overrides used only where the local spen

OpenCode-held OpenAI/Codex OAuth can be reused for **remote Codex account quota** only when the Codex provider's `External OAuth sources` setting is explicitly enabled. Native Codex credentials still take precedence, an explicit `CODEX_HOME` stays isolated, and external credentials remain read-only. This does **not** import ordinary OpenCode sessions into Codex token or spend totals. OpenCode Go's local SQLite reader remains scoped to its own `opencode-go` assistant records; OpenAI API-platform usage is a separate provider.

Codex local cost prices **Priority (Fast) turns** at the Fast rate. A turn counts as Priority when `<CODEX_HOME>/logs_2.sqlite` (Codex's trace database) holds a `response.create` websocket request with `service_tier == "priority"` for that turn's id. The database is opened read-only, scanned incrementally with a persisted cursor in the cost cache, and only turn ids, model names, and timestamps are kept; row bodies contain prompts and are never stored or logged. A missing or unreadable database keeps Standard pricing, and turn evidence never crosses `CODEX_HOME` scopes. Models without a Fast lane stay Standard. Older cost caches rebuild once (Codex cache schema v4 records each row's turn id).

### z.ai Coding Plan quotas

z.ai Coding Plans accept both `TOKENS_LIMIT` and `CREDIT_LIMIT` rows. The shortest known Coding Plan window becomes primary and the longest becomes secondary; `TIME_LIMIT` is the separate MCP lane. When absolute usage/remaining counts are available they determine the used percentage, otherwise the provider percentage is used, always clamped to 0–100%. This behavior is shared by the tray, provider detail, CLI, and other Windows surfaces.
Expand Down
52 changes: 17 additions & 35 deletions rust/src/codex_costs.rs
Original file line number Diff line number Diff line change
Expand Up @@ -361,27 +361,17 @@ fn add_codex_tokens_to_summary(
return Some(0.0);
}

let priced = pricing_day
.and_then(|day| {
CostUsagePricing::codex_cost_usd_at_date(
&model_key,
tokens.input,
tokens.cached,
tokens.output,
day,
)
})
.or_else(|| {
CostUsagePricing::codex_cost_usd(&model_key, tokens.input, tokens.cached, tokens.output)
});
let uses_fallback_pricing = priced.is_none();
let cost = codex_cost_usd_for_day(
let priced = CostUsagePricing::codex_day_aggregate_cost_usd(
&model_key,
tokens.input,
tokens.cached,
tokens.output,
pricing_day,
);
let uses_fallback_pricing = priced.is_none();
let cost = priced.unwrap_or_else(|| {
codex_cost_usd_fallback(&model_key, tokens.input, tokens.cached, tokens.output)
});
if uses_fallback_pricing {
summary.unknown_models.insert(model_key.clone());
match &mut summary.model_pricing_completeness {
Expand Down Expand Up @@ -484,26 +474,8 @@ fn codex_cost_usd_for_day(
if CostUsagePricing::is_codex_unattributed_model(model) {
return 0.0;
}
let priced = pricing_day
.and_then(|day| CostUsagePricing::codex_cost_usd_at_date(model, input, cached, output, day))
.or_else(|| CostUsagePricing::codex_cost_usd(model, input, cached, output));
if let Some(cost) = priced {
return cost;
}

let normalized = CostUsagePricing::normalize_codex_model(model);
if normalized.contains("fast") || normalized.contains("priority") {
let fast = pricing_day
.and_then(|day| {
CostUsagePricing::codex_fast_cost_usd_at_date(model, input, cached, output, day)
})
.or_else(|| CostUsagePricing::codex_fast_cost_usd(model, input, cached, output));
if let Some(cost) = fast {
return cost;
}
}

codex_cost_usd_fallback(model, input, cached, output)
CostUsagePricing::codex_day_aggregate_cost_usd(model, input, cached, output, pricing_day)
.unwrap_or_else(|| codex_cost_usd_fallback(model, input, cached, output))
}

fn codex_cost_usd_fallback(model: &str, input: u64, cached: u64, output: u64) -> f64 {
Expand Down Expand Up @@ -581,6 +553,7 @@ mod tests {
cached: 0,
output: 20,
reasoning,
turn_id: None,
};

let mut known_summary = CostSummary::default();
Expand Down Expand Up @@ -612,6 +585,7 @@ mod tests {
cached: 0,
output: 20,
reasoning,
turn_id: None,
};
let records = vec![
(make_record(Some(7)), 0),
Expand Down Expand Up @@ -675,6 +649,7 @@ mod tests {
cached: 0,
output: 0,
reasoning: None,
turn_id: None,
},
0,
),
Expand All @@ -687,6 +662,7 @@ mod tests {
cached: 0,
output: 0,
reasoning: None,
turn_id: None,
},
0,
),
Expand All @@ -699,6 +675,7 @@ mod tests {
cached: 0,
output: 0,
reasoning: None,
turn_id: None,
},
0,
),
Expand Down Expand Up @@ -750,6 +727,7 @@ mod tests {
cached: 0,
output: 5,
reasoning: None,
turn_id: None,
},
0,
),
Expand All @@ -762,6 +740,7 @@ mod tests {
cached: 0,
output: 1_000_000,
reasoning: None,
turn_id: None,
},
0,
),
Expand Down Expand Up @@ -789,6 +768,7 @@ mod tests {
cached: 0,
output: 1,
reasoning: None,
turn_id: None,
},
0,
)];
Expand All @@ -810,6 +790,7 @@ mod tests {
cached: 0,
output: 0,
reasoning: None,
turn_id: None,
},
0,
)];
Expand Down Expand Up @@ -843,6 +824,7 @@ mod tests {
cached: 0,
output: 1_000_000,
reasoning: None,
turn_id: None,
},
0,
)];
Expand Down
42 changes: 20 additions & 22 deletions rust/src/codex_costs/quota_windows.rs
Original file line number Diff line number Diff line change
Expand Up @@ -7,9 +7,11 @@
use chrono::{DateTime, Duration, Local, NaiveDate, TimeZone, Utc};
use serde::{Deserialize, Serialize};
use std::collections::HashSet;
use std::path::Path;

use crate::core::{
CodexSourceRowCache, CodexSourceUsageRow, CostUsageCache, CostUsagePricing, RateWindow,
CodexPriorityOverlay, CodexSourceRowCache, CodexSourceUsageRow, CostUsageCache,
CostUsagePricing, RateWindow, row_priced_model,
};

const NOMINAL_WEEK_MINUTES: i64 = 7 * 24 * 60;
Expand Down Expand Up @@ -341,7 +343,9 @@ fn cache_slices(cache: &CostUsageCache) -> Vec<Slice> {
});
let mut identities = HashSet::new();
let mut slices = Vec::new();
let cursor = cache.codex_priority_turns_cursor.as_ref();
for (path, source) in sources {
let overlay = cursor.and_then(|cursor| cursor.overlay_for_file(Path::new(path)));
let identity = if source.file_identity.is_empty() {
path.as_str()
} else {
Expand All @@ -351,37 +355,29 @@ fn cache_slices(cache: &CostUsageCache) -> Vec<Slice> {
continue;
}
for row in &source.rows {
slices.push(slice_from_row(row));
slices.push(slice_from_row(row, overlay.as_ref()));
}
}
slices.sort_by_key(|slice| (slice.start, slice.end));
slices
}

fn slice_from_row(row: &CodexSourceUsageRow) -> Slice {
fn slice_from_row(row: &CodexSourceUsageRow, overlay: Option<&CodexPriorityOverlay<'_>>) -> Slice {
let timestamp = row.timestamp.or_else(|| local_day_start(&row.day_key));
let end = row.timestamp.map(|_| None).unwrap_or_else(|| {
local_day_start(&row.day_key).and_then(|start| start.checked_add_signed(Duration::days(1)))
});
let input = u64::try_from(row.input.max(0)).unwrap_or(0);
let output = u64::try_from(row.output.max(0)).unwrap_or(0);
let tokens = Some(input.saturating_add(output));
let cost_usd = row.pricing.pricing_model.as_deref().and_then(|model| {
let model = if row.pricing.pricing_mode.as_deref() == Some("priority")
&& !model.ends_with("-priority")
{
format!("{model}-priority")
} else {
model.to_string()
};
let cost_usd = row_priced_model(row, overlay).and_then(|model| {
let date = timestamp.map(|value| value.with_timezone(&Local).date_naive())?;
CostUsagePricing::codex_cost_usd_at_date(
&model,
input,
u64::try_from(row.cached.max(0)).unwrap_or(0).min(input),
output,
date,
)
let cached = u64::try_from(row.cached.max(0)).unwrap_or(0).min(input);
if model.ends_with("-priority") {
CostUsagePricing::codex_fast_cost_usd_at_date(&model, input, cached, output, date)
} else {
CostUsagePricing::codex_cost_usd_at_date(&model, input, cached, output, date)
}
});
Slice {
start: timestamp.unwrap_or(DateTime::<Utc>::UNIX_EPOCH),
Expand Down Expand Up @@ -418,14 +414,16 @@ fn legacy_day_slices(cache: &CostUsageCache) -> Vec<Slice> {
let cached = u64::try_from(packed.get(1).copied().unwrap_or(0).max(0))
.unwrap_or(0)
.min(input);
let cost_usd = CostUsagePricing::codex_cost_usd_at_date(
let cost_usd = CostUsagePricing::codex_day_aggregate_cost_usd(
model,
input,
cached,
output,
NaiveDate::parse_from_str(day, "%Y-%m-%d")
.ok()
.unwrap_or_else(|| start.with_timezone(&Local).date_naive()),
Some(
NaiveDate::parse_from_str(day, "%Y-%m-%d")
.ok()
.unwrap_or_else(|| start.with_timezone(&Local).date_naive()),
),
);
slices.push(Slice {
start,
Expand Down
1 change: 1 addition & 0 deletions rust/src/codex_costs/quota_windows/tests.rs
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,7 @@ fn row(
output,
reasoning: None,
source_end_offset: 1,
turn_id: None,
pricing: CodexSourcePricingEvidence {
pricing_model: pricing_model.map(str::to_string),
pricing_mode: None,
Expand Down
103 changes: 103 additions & 0 deletions rust/src/core/cost_pricing/codex.rs
Original file line number Diff line number Diff line change
@@ -1,3 +1,5 @@
use chrono::NaiveDate;

use super::super::{codex_routed_pricing, models_dev_pricing};
use super::{CODEX_PRICING, CostUsagePricing};

Expand Down Expand Up @@ -60,6 +62,107 @@ pub(super) fn codex_fast_allows_long_context(model: &str) -> bool {
}

impl CostUsagePricing {
/// Whether one request of `input_tokens` can run in the Fast lane of
/// `model`. Older models offer no Fast lane above the long-context
/// threshold, so upstream charges such a Priority request the Standard
/// cost; Astra publishes long-context Fast rates.
pub fn codex_fast_lane_covers(model: &str, input_tokens: u64) -> bool {
Self::codex_api_fast_multiplier(model).is_some()
&& (input_tokens <= CODEX_LONG_CONTEXT_THRESHOLD
|| codex_fast_allows_long_context(model))
}

/// Fast cost in USD of a day aggregate under a Fast key (`-priority` or
/// `-fast`), or `None` when `model` names no Fast lane.
///
/// Upstream prices every request on its own. A day aggregate sums
/// requests that each ran in the Fast lane, so the summed input must
/// neither refuse the surcharge nor switch to long-context rates: older
/// models price at the base model's short-context rates times the
/// multiplier. Astra's Fast lane has long-context rates, so its
/// aggregate keeps the whole-aggregate rule that Standard aggregates use.
pub fn codex_fast_aggregate_cost_usd(
model: &str,
input_tokens: u64,
cached_input_tokens: u64,
output_tokens: u64,
pricing_date: Option<NaiveDate>,
) -> Option<f64> {
let base = Self::codex_fast_base_model(model);
if base == Self::normalize_codex_model(model) {
return None;
}
if codex_fast_allows_long_context(model) {
return pricing_date
.and_then(|date| {
Self::codex_fast_cost_usd_at_date(
model,
input_tokens,
cached_input_tokens,
output_tokens,
date,
)
})
.or_else(|| {
Self::codex_fast_cost_usd(
model,
input_tokens,
cached_input_tokens,
output_tokens,
)
});
}
let multiplier = Self::codex_api_fast_multiplier(model)?;
let (input_rate, cache_read_rate, output_rate) =
Self::codex_short_context_rates(&base, pricing_date)?;
Some(
codex_cost_from_rates(
input_tokens,
cached_input_tokens,
output_tokens,
input_rate,
cache_read_rate,
output_rate,
) * multiplier,
)
}

/// Known cost in USD of one Codex day aggregate: a Fast key prices
/// through its base model's Fast lane
/// ([`Self::codex_fast_aggregate_cost_usd`]), any other model at the
/// rates in effect on `pricing_date`. `None` means no rate is known.
pub fn codex_day_aggregate_cost_usd(
model: &str,
input_tokens: u64,
cached_input_tokens: u64,
output_tokens: u64,
pricing_date: Option<NaiveDate>,
) -> Option<f64> {
let (input, cached, output) = (input_tokens, cached_input_tokens, output_tokens);
Self::codex_fast_aggregate_cost_usd(model, input, cached, output, pricing_date)
.or_else(|| {
pricing_date.and_then(|date| {
Self::codex_cost_usd_at_date(model, input, cached, output, date)
})
})
.or_else(|| Self::codex_cost_usd(model, input, cached, output))
}

/// Short-context `(input, cache read, output)` rates of `model` on
/// `pricing_date`. Short-context pricing is linear per token, so
/// one-token probes read the exact dated rates back.
fn codex_short_context_rates(
model: &str,
pricing_date: Option<NaiveDate>,
) -> Option<(f64, f64, f64)> {
let cost = |input, cached, output| {
pricing_date
.and_then(|date| Self::codex_cost_usd_at_date(model, input, cached, output, date))
.or_else(|| Self::codex_cost_usd(model, input, cached, output))
};
Some((cost(1, 0, 0)?, cost(1, 1, 0)?, cost(0, 0, 1)?))
}

/// Calculate Codex cost in USD when input includes cache-write tokens.
pub fn codex_cost_usd_with_cache_write(
model: &str,
Expand Down
Loading