diff --git a/README.md b/README.md index b26314a..fe5cba5 100644 --- a/README.md +++ b/README.md @@ -90,7 +90,7 @@ Click the menu-bar item to open a three-tab panel: - **Composition** (构成) — where the money went: spend broken down by model and by project. - **Insights** (洞察) — code output, tool-use mix, golden-hours heatmap, savings tips, and a quota-depletion forecast. -> **How the numbers are computed.** *Spend* is an **estimate at pay-as-you-go API prices** (`Pricing.swift`), **not your subscription bill** — on a Max / ChatGPT plan, read it as "equivalent API value," not money actually charged. *Code output* is **approximate git attribution**: all non-merge commits in a session's working directory within the time window — it can't tell hand-written from AI commits and excludes uncommitted work. *Codex* token totals are de-duplicated from each session's cumulative counter (`total_token_usage`), so they no longer double-count replayed events. +> **How the numbers are computed.** *Spend* is an **estimate at standard pay-as-you-go API prices** (`Pricing.swift`), **not your subscription bill** — on a Max / ChatGPT plan, read it as "equivalent API value," not money actually charged. Models without a published rate use a visibly approximate estimate. *Code output* is **approximate git attribution**: all non-merge commits in a session's working directory within the time window — it can't tell hand-written from AI commits and excludes uncommitted work. *Codex* uses per-response token usage in recent logs; older logs use cumulative-counter deltas without charging the prior session's starting balance. ## Privacy diff --git a/README_ZH.md b/README_ZH.md index 4fb2609..acbc5b0 100644 --- a/README_ZH.md +++ b/README_ZH.md @@ -89,7 +89,7 @@ make package # 产出 dist/CodingBar.app - **构成**(Composition)— 钱花在哪:按模型和按项目拆解花费。 - **洞察**(Insights)— 代码产出、工具使用占比、黄金时段热力图、省钱提示、额度燃尽预测。 -> **数字是怎么算的。** *花费*是按 **API 现付价估算**(`Pricing.swift`),**不是你的订阅账单**——包月 Max / ChatGPT 用户应把它读作「等价 API 价值」,而非实际扣款。*代码产出*是**近似的 git 归因**:会话工作目录在时间窗内的全部非 merge 提交——无法区分手写与 AI 提交,且不含未提交改动。*Codex* 的 token 总量按每个会话的累计计数器(`total_token_usage`)做差分去重,不再把重复事件算两遍。 +> **数字是怎么算的。** *花费*是按 **标准 API 现付价估算**(`Pricing.swift`),**不是你的订阅账单**——包月 Max / ChatGPT 用户应把它读作「等价 API 价值」,而非实际扣款。未公布价格的模型使用带近似标记的估价。*代码产出*是**近似的 git 归因**:会话工作目录在时间窗内的全部非 merge 提交——无法区分手写与 AI 提交,且不含未提交改动。*Codex* 的新日志使用逐请求 token 用量;旧日志从累计计数器做差分,不把前一会话的初始累计值算作本次用量。 ## 隐私 diff --git a/Sources/CodingBar/SelfTest.swift b/Sources/CodingBar/SelfTest.swift index 0eebd5c..3633ac6 100644 --- a/Sources/CodingBar/SelfTest.swift +++ b/Sources/CodingBar/SelfTest.swift @@ -30,10 +30,16 @@ enum SelfTest { let september = Date(timeIntervalSince1970: 1_788_220_800) check("Fable 5 1h cache pricing", abs(Pricing.cost(model: "claude-fable-5", tokens: millionTokens, at: july, cacheWrite1h: 1_000_000) - 81) < 0.000_001) - check("Sonnet 5 intro pricing", abs(Pricing.cost(model: "claude-sonnet-5", tokens: millionTokens, - at: july, cacheWrite1h: 1_000_000) - 16.2) < 0.000_001) - check("Sonnet 5 standard pricing", abs(Pricing.cost(model: "claude-sonnet-5", tokens: millionTokens, - at: september, cacheWrite1h: 1_000_000) - 24.3) < 0.000_001) + check("Sonnet 5 permanent price", abs(Pricing.cost(model: "claude-sonnet-5", tokens: millionTokens, + at: july, cacheWrite1h: 1_000_000) - 16.2) < 0.000_001 + && abs(Pricing.cost(model: "claude-sonnet-5", tokens: millionTokens, + at: september, cacheWrite1h: 1_000_000) - 16.2) < 0.000_001) + check("latest Claude tiers and cache reads", + abs(Pricing.cost(model: "claude-opus-5-5", tokens: millionTokens, + at: september, cacheWrite1h: 1_000_000) - 32.2) < 0.000_001 + && abs(Pricing.cost(model: "claude-fable-5-1", tokens: millionTokens, + at: september, cacheWrite1h: 1_000_000) - 80.25) < 0.000_001 + && Pricing.priceIsExact(model: "claude-mythos-5-1")) let openAIBaseTokens = TokenBreakdown(input: 100_000, output: 100_000, cacheRead: 100_000, cacheWrite: 100_000) @@ -44,10 +50,18 @@ enum SelfTest { && Pricing.normalize(model: "gpt-5.6-luna") == "openai/gpt-5.6-luna" && Pricing.priceIsExact(model: "gpt-5.6")) check("GPT-5.6 Sol base and long-context pricing", - abs(Pricing.cost(model: "gpt-5.6-sol", tokens: openAIBaseTokens, at: july, - billingInputTokens: 272_000) - 4.175) < 0.000_001 - && abs(Pricing.cost(model: "gpt-5.6-sol", tokens: openAIBaseTokens, at: july, - billingInputTokens: 272_001) - 6.85) < 0.000_001) + abs(Pricing.cost(model: "gpt-5.6-sol", tokens: openAIBaseTokens, at: september, + billingInputTokens: 272_000) - 2.94) < 0.000_001 + && abs(Pricing.cost(model: "gpt-5.6-sol", tokens: openAIBaseTokens, at: september, + billingInputTokens: 272_001) - 4.88) < 0.000_001) + check("GPT-6 exact rates and long context", + Pricing.priceIsExact(model: "gpt-6-astra") + && Pricing.priceIsExact(model: "gpt-6-sol") + && Pricing.priceIsExact(model: "gpt-6-luna") + && abs(Pricing.cost(model: "gpt-6-astra", tokens: openAIBaseTokens, + at: september, billingInputTokens: 272_001) - 12.2) < 0.000_001 + && abs(Pricing.cost(model: "gpt-6-luna", tokens: openAIBaseTokens, + at: september, billingInputTokens: 272_001) - 0.122) < 0.000_001) check("GPT prices cover current and historical IDs", abs(Pricing.cost(model: "gpt-5.4-mini", tokens: TokenBreakdown(output: 1_000_000), at: july) - 4.5) < 0.000_001 @@ -63,22 +77,20 @@ enum SelfTest { // Regression: the family-keyword fallback used to funnel every Opus into 4.8, so a // real `claude-opus-5` record was renamed and merged into the 4.8 row. Each tier - // must resolve to itself, and an unrecognized version to the newest — not a pinned - // older one, which is how this broke in the first place. + // must resolve to itself; an unrecognized version keeps its own approximate ID. check("Opus 5 keeps its own identity", Pricing.normalize(model: "claude-opus-5") == "anthropic/claude-opus-5" && Pricing.displayName(forCanonicalKey: "anthropic/claude-opus-5") == "Opus 5") check("older Opus tiers still resolve to themselves", Pricing.normalize(model: "claude-opus-4-8") == "anthropic/claude-opus-4-8" && Pricing.normalize(model: "claude-opus-4-6") == "anthropic/claude-opus-4-6") - check("dated Opus 5 variant resolves via the fallback", + check("dated Opus 5 variant resolves exactly", Pricing.normalize(model: "claude-opus-5-20260315") == "anthropic/claude-opus-5") - check("unknown Opus/Sonnet versions resolve to the newest, not a pinned tier", - Pricing.normalize(model: "claude-opus-9") == "anthropic/claude-opus-5" - && Pricing.normalize(model: "claude-sonnet-9") == "anthropic/claude-sonnet-5") - // "Unknown → newest" is only safe while every *known* tier is enumerated: Opus 4.1 - // costs 3x the 4.5+ tiers, so falling through to Opus 5 would bill it at a third of - // its real rate. Mythos 5 is Fable-tier and would otherwise hit the $3/$15 fallback. + check("unknown Claude versions keep their identity and approximate marker", + Pricing.normalize(model: "claude-opus-9") == "claude-opus-9" + && !Pricing.priceIsExact(model: "claude-opus-9") + && Pricing.normalize(model: "claude-sonnet-9") == "claude-sonnet-9") + // Opus 4.1 costs 3x the 4.5+ tiers, so explicit historical rows matter. check("off-tier Opus versions resolve to themselves, not the newest", Pricing.normalize(model: "claude-opus-4-1-20250805") == "anthropic/claude-opus-4-1" && Pricing.normalize(model: "claude-opus-4-5-20251101") == "anthropic/claude-opus-4-5") @@ -89,7 +101,7 @@ enum SelfTest { && abs(Pricing.cost(model: "claude-mythos-5", tokens: millionTokens, at: july) - Pricing.cost(model: "claude-fable-5", tokens: millionTokens, at: july)) < 0.000_001) check("bare family selectors mean the current model", - Pricing.normalize(model: "opus") == "anthropic/claude-opus-5" + Pricing.normalize(model: "opus") == "anthropic/claude-opus-5-5" && Pricing.normalize(model: "sonnet") == "anthropic/claude-sonnet-5") // 1M each of input/output/cacheRead/cacheWrite at $5 / $25 / $0.5 / $6.25 = $36.75. check("Opus 5 priced at the Opus tier, not the generic fallback", diff --git a/Sources/CodingBarCore/Aggregator.swift b/Sources/CodingBarCore/Aggregator.swift index c0c40d6..794aab1 100644 --- a/Sources/CodingBarCore/Aggregator.swift +++ b/Sources/CodingBarCore/Aggregator.swift @@ -144,18 +144,20 @@ public enum Aggregator { // contributing record priced via a family guess / fallback rate. let home = FileManager.default.homeDirectoryForCurrentUser.path func breakdown(from records: [RawRecord]) -> (models: [ModelStat], projects: [ProjectStat]) { - var modelMap: [String: (tokens: TokenBreakdown, cost: Double, exact: Bool)] = [:] + var modelMap: [String: (model: String, provider: Provider, tokens: TokenBreakdown, cost: Double, exact: Bool)] = [:] for r in records { let key = Pricing.normalize(model: r.model) - var entry = modelMap[key] ?? (tokens: TokenBreakdown(), cost: 0, exact: true) + let groupKey = r.provider.rawValue + "·" + key + var entry = modelMap[groupKey] ?? (model: key, provider: r.provider, + tokens: TokenBreakdown(), cost: 0, exact: true) entry.tokens += r.tokens entry.cost += recordCost(r) entry.exact = entry.exact && Pricing.priceIsExact(model: r.model) - modelMap[key] = entry + modelMap[groupKey] = entry } let models: [ModelStat] = modelMap - .map { key, entry in - ModelStat(model: key, provider: Pricing.provider(forCanonicalKey: key), + .map { _, entry in + ModelStat(model: entry.model, provider: entry.provider, tokens: entry.tokens, cost: entry.cost, pricedExact: entry.exact) } .sorted { $0.cost > $1.cost } diff --git a/Sources/CodingBarCore/ClaudeScanner.swift b/Sources/CodingBarCore/ClaudeScanner.swift index 53a809b..7b144c1 100644 --- a/Sources/CodingBarCore/ClaudeScanner.swift +++ b/Sources/CodingBarCore/ClaudeScanner.swift @@ -17,21 +17,31 @@ public enum ClaudeScanner { return ([], []) } - var seenIds = Set() - var allRecords: [RawRecord] = [] - let records = scanner.scan(directory: projectsDir) { fileURL in parseFile(fileURL) } + return deduplicate(records) + } - // Dedup by message.id across all files + /// Streamed assistant messages repeat the same id. The later record carries + /// the final output-token count and often the completed tool-use content. + static func deduplicate(_ records: [RawRecord]) -> (records: [RawRecord], seenIds: Set) { + var seenIds = Set() + var indexByID: [String: Int] = [:] + var allRecords: [RawRecord] = [] for record in records { if let mid = record.messageId { - guard seenIds.insert(mid).inserted else { continue } + if let index = indexByID[mid] { + if record.tokens.output >= allRecords[index].tokens.output { + allRecords[index] = record + } + continue + } + indexByID[mid] = allRecords.count + seenIds.insert(mid) } allRecords.append(record) } - return (allRecords, seenIds) } diff --git a/Sources/CodingBarCore/Coach.swift b/Sources/CodingBarCore/Coach.swift index ccecbf9..996ab97 100644 --- a/Sources/CodingBarCore/Coach.swift +++ b/Sources/CodingBarCore/Coach.swift @@ -2,20 +2,6 @@ import Foundation enum Coach { - // Canonical keys for Opus and Haiku pricing families. Every Opus tier belongs here: - // a missing one doesn't degrade the tip, it silently excludes that model's turns from - // the count entirely, so the advice goes quiet exactly when a new Opus becomes the - // model people actually run. - private static let opusKeys: Set = [ - "anthropic/claude-opus-5", - "anthropic/claude-opus-4-8", - "anthropic/claude-opus-4-7", - "anthropic/claude-opus-4-6", - ] - private static let haikuKeys: Set = [ - "anthropic/claude-haiku-4-5", - ] - // A "simple" turn has zero or one tool call and fewer than 300 output tokens. private static func isSimpleTurn(_ record: RawRecord) -> Bool { record.toolNames.count <= 1 && record.tokens.output < 300 @@ -24,43 +10,24 @@ enum Coach { static func opusOnSimpleTip(from todayRecords: [RawRecord], language: AppLanguage) -> Insight? { let claudeToday = todayRecords.filter { $0.provider == .claude } - // Cost delta: only count the non-cached input (cache tokens are already cheap - // regardless of model — switching models won't help much there). - var totalSimpleNetInput = 0 // non-cached input tokens only - var totalSimpleCacheRead = 0 // cache-read tokens (priced differently) - var totalSimpleOutput = 0 + // Compare each turn at its actual Opus rate; cache writes are excluded + // because switching models would recreate the cache rather than reuse it. + var totalSaved = 0.0 var count = 0 for r in claudeToday { let key = Pricing.normalize(model: r.model) - guard opusKeys.contains(key) else { continue } + guard key.hasPrefix("anthropic/claude-opus-"), Pricing.priceIsExact(model: r.model) else { continue } guard isSimpleTurn(r) else { continue } - totalSimpleNetInput += r.tokens.input - totalSimpleCacheRead += r.tokens.cacheRead - totalSimpleOutput += r.tokens.output + let compared = TokenBreakdown(input: r.tokens.input, output: r.tokens.output, + cacheRead: r.tokens.cacheRead) + totalSaved += Pricing.cost(model: r.model, tokens: compared, at: r.timestamp) + - Pricing.cost(model: "claude-haiku-4-5", tokens: compared, at: r.timestamp) count += 1 } guard count >= 3 else { return nil } // not enough to matter - // Price the delta off the current Opus, not a pinned older one. Identical numbers - // today (both tiers are $5/$25), but this is what keeps the saving honest the next - // time the tiers diverge. - let opusKey = "anthropic/claude-opus-5" - let haikuKey = "anthropic/claude-haiku-4-5" - let opusInputPrice = Pricing.inputPrice(forCanonicalKey: opusKey) - let haikuInputPrice = Pricing.inputPrice(forCanonicalKey: haikuKey) - // Cache read price delta is small; include it for completeness - let opusCacheReadPrice = Pricing.cacheReadPrice(forCanonicalKey: opusKey) - let haikuCacheReadPrice = Pricing.cacheReadPrice(forCanonicalKey: haikuKey) - let opusOutputPricePerM = 25.0 // USD/1M - let haikuOutputPricePerM = 5.0 // USD/1M - - let savedInput = Double(totalSimpleNetInput) * (opusInputPrice - haikuInputPrice) / 1_000_000 - let savedCacheRead = Double(totalSimpleCacheRead) * (opusCacheReadPrice - haikuCacheReadPrice) / 1_000_000 - let savedOutput = Double(totalSimpleOutput) * (opusOutputPricePerM - haikuOutputPricePerM) / 1_000_000 - let totalSaved = savedInput + savedCacheRead + savedOutput - guard totalSaved >= 0.2 else { return nil } let saved = String(format: "%.2f", totalSaved) diff --git a/Sources/CodingBarCore/CodexScanner.swift b/Sources/CodingBarCore/CodexScanner.swift index a5712fc..f06ad1b 100644 --- a/Sources/CodingBarCore/CodexScanner.swift +++ b/Sources/CodingBarCore/CodexScanner.swift @@ -17,9 +17,14 @@ public enum CodexScanner { return [] } - return scanner.scan(directory: sessionsDir) { fileURL in + let records = scanner.scan(directory: sessionsDir) { fileURL in parseFile(fileURL) } + var seenResponses = Set() + return records.filter { record in + guard let id = record.messageId else { return true } + return seenResponses.insert(id).inserted + } } static func parseFile(_ fileURL: URL) -> [RawRecord] { @@ -36,15 +41,16 @@ public enum CodexScanner { var cwd = "" var model = "unknown" var records: [RawRecord] = [] - // Codex `token_count` events carry a CUMULATIVE `total_token_usage` snapshot - // that grows every turn. We used to sum the per-turn `last_token_usage`, but - // replayed/duplicate events inflated that sum past the session's real total - // (measured ~1.3–1.8× across this machine's logs). Taking the positive delta - // of `total_token_usage` reconstructs each turn's true increment, drops - // duplicate snapshots (Δ≤0), and preserves per-turn timestamps for bucketing. - var prevInput = 0, prevCached = 0, prevCacheWrite = 0, prevOutput = 0, prevReasoning = 0 + // Modern rollouts write one token_usage_record per response, followed by a + // token_count snapshot of the same usage. Once the former appears, the latter + // must be ignored. Older rollouts only have token_count; their cumulative + // counter may already include earlier files, so the first request (or a reset) + // must come from last_token_usage, not the cumulative total. + var hasModernUsage = false + var seenResponses = Set() + var previousTotal: [String: Any]? // Codex tool calls (`function_call` response items, e.g. exec_command) arrive - // before the turn's `token_count`; buffer their names and attach them to the + // before the turn's usage record; buffer their names and attach them to the // next emitted record so the habits tool-mix counts Codex, not just Claude. var pendingTools: [String] = [] @@ -88,43 +94,60 @@ public enum CodexScanner { pendingTools.append(name) } + case "token_usage_record": + guard let payload = obj["payload"] as? [String: Any], + let usage = payload["usage"] as? [String: Any] else { return } + let responseID = (payload["response_id"] as? String).flatMap { $0.isEmpty ? nil : $0 } + if let responseID, seenResponses.contains(responseID) { + pendingTools.removeAll(keepingCapacity: true) + return + } + guard let timestamp = iso.date(from: obj["timestamp"] as? String) else { + pendingTools.removeAll(keepingCapacity: true) + return + } + guard let record = makeRecord(usage: usage, billingInputTokens: usage["input_tokens"] as? Int, + timestamp: timestamp, responseID: responseID) else { + pendingTools.removeAll(keepingCapacity: true) + return + } + if let responseID { _ = seenResponses.insert(responseID) } + hasModernUsage = true + records.append(record) + pendingTools.removeAll(keepingCapacity: true) + case "event_msg": guard let payload = obj["payload"] as? [String: Any], let payloadType = payload["type"] as? String, payloadType == "token_count" else { return } - - // Every non-null `info` carries `total_token_usage` (verified across - // every real event). Its positive delta remains the billable token count; - // `last_token_usage.input_tokens` is retained only as the absolute prompt - // size needed to select OpenAI's >272K long-context price tier. + if hasModernUsage { return } guard let info = payload["info"] as? [String: Any], let total = info["total_token_usage"] as? [String: Any] else { return } let last = info["last_token_usage"] as? [String: Any] - let billingInputTokens = (last?["input_tokens"] as? Int).map { max(0, $0) } - - let curInput = total["input_tokens"] as? Int ?? 0 - let curCached = total["cached_input_tokens"] as? Int ?? 0 - let curCacheWrite = total["cache_write_input_tokens"] as? Int ?? 0 - let curOutput = total["output_tokens"] as? Int ?? 0 - let curReasoning = total["reasoning_output_tokens"] as? Int ?? 0 - - // Δ of the cumulative counter. A counter that *drops* (post-compaction - // reset) starts a fresh baseline so those turns aren't lost. - let reset = curInput < prevInput || curOutput < prevOutput - let dInput = reset ? curInput : curInput - prevInput - let dCached = reset ? curCached : curCached - prevCached - let dCacheWrite = reset ? curCacheWrite : curCacheWrite - prevCacheWrite - let dOutput = reset ? curOutput : curOutput - prevOutput - let dReasoning = reset ? curReasoning : curReasoning - prevReasoning - prevInput = curInput; prevCached = curCached; prevCacheWrite = curCacheWrite - prevOutput = curOutput; prevReasoning = curReasoning - - // No forward progress → a replayed/duplicate snapshot, nothing billed. - guard dInput + dOutput > 0 else { return } + let keys = ["input_tokens", "cached_input_tokens", "cache_write_input_tokens", + "output_tokens", "reasoning_output_tokens"] + let reset = previousTotal.map { previous in + keys.contains { (total[$0] as? Int ?? 0) < (previous[$0] as? Int ?? 0) } + } ?? true + let usage: [String: Any]? + if reset { + usage = last + } else { + var delta: [String: Any] = [:] + for key in keys { + delta[key] = (total[key] as? Int ?? 0) - (previousTotal?[key] as? Int ?? 0) + } + usage = delta + } + previousTotal = total + guard let usage, + (usage["input_tokens"] as? Int ?? 0) + (usage["output_tokens"] as? Int ?? 0) > 0 else { + return + } // Unparseable/absent timestamp → DROP rather than fall back to Date(). // Clear this turn's buffered tools too (they belong to the dropped record, @@ -134,39 +157,10 @@ public enum CodexScanner { pendingTools.removeAll(keepingCapacity: true); return } - // Codex input_tokens includes both cached reads and cache writes. Keep all - // three buckets disjoint so both total tokens and model-specific cache rates - // remain correct. Negative subset deltas are treated as zero after a reset. - let cacheRead = max(0, dCached) - let cacheWrite = max(0, dCacheWrite) - let netInput = max(0, dInput - cacheRead - cacheWrite) - - // Codex's output_tokens already includes reasoning_output_tokens. Split - // the subset into its own bucket so TokenBreakdown.total and Pricing.cost - // count it once rather than adding the same reasoning tokens twice. - let reasoning = min(max(0, dReasoning), max(0, dOutput)) - let tokens = TokenBreakdown( - input: netInput, - output: max(0, dOutput - reasoning), - cacheRead: cacheRead, - cacheWrite: cacheWrite, - reasoning: reasoning - ) - - let record = RawRecord( - provider: .codex, - model: model, - timestamp: timestamp, - cwd: cwd, - tokens: tokens, - billingInputTokens: billingInputTokens, - toolName: pendingTools.first, - toolNames: pendingTools, - messageId: nil, - sessionKey: sessionKey, - hasInterrupt: false - ) - records.append(record) + if let record = makeRecord(usage: usage, billingInputTokens: last?["input_tokens"] as? Int, + timestamp: timestamp, responseID: nil) { + records.append(record) + } pendingTools.removeAll(keepingCapacity: true) default: @@ -176,5 +170,22 @@ public enum CodexScanner { } return records + + func makeRecord(usage: [String: Any], billingInputTokens: Int?, timestamp: Date, + responseID: String?) -> RawRecord? { + let totalInput = max(0, usage["input_tokens"] as? Int ?? 0) + let cacheRead = max(0, usage["cached_input_tokens"] as? Int ?? 0) + let cacheWrite = max(0, usage["cache_write_input_tokens"] as? Int ?? 0) + let totalOutput = max(0, usage["output_tokens"] as? Int ?? 0) + guard totalInput + totalOutput > 0 else { return nil } + let reasoning = min(max(0, usage["reasoning_output_tokens"] as? Int ?? 0), totalOutput) + let tokens = TokenBreakdown(input: max(0, totalInput - cacheRead - cacheWrite), + output: totalOutput - reasoning, cacheRead: cacheRead, + cacheWrite: cacheWrite, reasoning: reasoning) + return RawRecord(provider: .codex, model: model, timestamp: timestamp, cwd: cwd, + tokens: tokens, billingInputTokens: billingInputTokens.map { max(0, $0) }, + toolName: pendingTools.first, toolNames: pendingTools, + messageId: responseID, sessionKey: sessionKey, hasInterrupt: false) + } } } diff --git a/Sources/CodingBarCore/Fuel.swift b/Sources/CodingBarCore/Fuel.swift index 3432b86..6da976a 100644 --- a/Sources/CodingBarCore/Fuel.swift +++ b/Sources/CodingBarCore/Fuel.swift @@ -221,7 +221,7 @@ enum FuelCalculator { sessions.append(LiveSession( name: name, model: Pricing.displayName(forCanonicalKey: mkey), - provider: Pricing.provider(forCanonicalKey: mkey), + provider: .claude, usedTokens: used, maxTokens: maxTok, throughput: tput diff --git a/Sources/CodingBarCore/Models.swift b/Sources/CodingBarCore/Models.swift index 85eed81..e379570 100644 --- a/Sources/CodingBarCore/Models.swift +++ b/Sources/CodingBarCore/Models.swift @@ -33,7 +33,7 @@ public struct TokenBreakdown: Codable, Sendable, Equatable { } public struct ModelStat: Codable, Sendable, Identifiable { - public var id: String { model } + public var id: String { provider.rawValue + "·" + model } public var model: String public var provider: Provider public var tokens: TokenBreakdown diff --git a/Sources/CodingBarCore/Pricing.swift b/Sources/CodingBarCore/Pricing.swift index 0654909..438ae12 100644 --- a/Sources/CodingBarCore/Pricing.swift +++ b/Sources/CodingBarCore/Pricing.swift @@ -35,6 +35,7 @@ public enum Pricing { private static let priceTable: [String: ModelPrice] = [ // Anthropic Claude — official models + "anthropic/claude-opus-5-5": ModelPrice(input: 4, output: 20, cacheRead: 0.2, cacheWrite5m: 5, cacheWrite1h: 8), "anthropic/claude-opus-5": ModelPrice(input: 5, output: 25, cacheRead: 0.5, cacheWrite5m: 6.25, cacheWrite1h: 10), "anthropic/claude-opus-4-8": ModelPrice(input: 5, output: 25, cacheRead: 0.5, cacheWrite5m: 6.25, cacheWrite1h: 10), "anthropic/claude-opus-4-7": ModelPrice(input: 5, output: 25, cacheRead: 0.5, cacheWrite5m: 6.25, cacheWrite1h: 10), @@ -43,17 +44,23 @@ public enum Pricing { // Deprecated (retires 2026-08-05) but priced 3x the 4.5+ tiers, so it must be // enumerated: the "unknown Opus → newest" fallback would otherwise bill it at $5/$25. "anthropic/claude-opus-4-1": ModelPrice(input: 15, output: 75, cacheRead: 1.5, cacheWrite5m: 18.75, cacheWrite1h: 30), + "anthropic/claude-fable-5-1": ModelPrice(input: 10, output: 50, cacheRead: 0.25, cacheWrite5m: 12.5, cacheWrite1h: 20), "anthropic/claude-fable-5": ModelPrice(input: 10, output: 50, cacheRead: 1, cacheWrite5m: 12.5, cacheWrite1h: 20), // Project Glasswing, invitation-only — same tier as Fable 5. Without a row it would // land on the generic $3/$15 fallback, i.e. 3.3x underpriced. + "anthropic/claude-mythos-5-1": ModelPrice(input: 10, output: 50, cacheRead: 0.25, cacheWrite5m: 12.5, cacheWrite1h: 20), "anthropic/claude-mythos-5": ModelPrice(input: 10, output: 50, cacheRead: 1, cacheWrite5m: 12.5, cacheWrite1h: 20), - "anthropic/claude-sonnet-5": ModelPrice(input: 3, output: 15, cacheRead: 0.3, cacheWrite5m: 3.75, cacheWrite1h: 6), + "anthropic/claude-sonnet-5": ModelPrice(input: 2, output: 10, cacheRead: 0.2, cacheWrite5m: 2.5, cacheWrite1h: 4), "anthropic/claude-sonnet-4-6": ModelPrice(input: 3, output: 15, cacheRead: 0.3, cacheWrite5m: 3.75, cacheWrite1h: 6), "anthropic/claude-haiku-4-5": ModelPrice(input: 1, output: 5, cacheRead: 0.1, cacheWrite5m: 1.25, cacheWrite1h: 2), - // OpenAI pay-as-you-go rates, current 2026-08-09. Pro models publish no - // cached-input discount, so their cache reads are billed at the full input rate. - // GPT-5.6 also charges cache writes at 1.25x input; both TTL fields use that one tier. - "openai/gpt-5.6-sol": openAI(input: 5, cachedInput: 0.5, output: 30, cacheWrite: 6.25, longContext: true), + // OpenAI standard pay-as-you-go rates. Pro models publish no cached-input + // discount. GPT-6 and GPT-5.6 cache writes use 1.25x input; both TTL fields + // use that one tier. Source: https://developers.openai.com/api/docs/pricing + "openai/gpt-6-astra": openAI(input: 10, cachedInput: 1, output: 50, cacheWrite: 12.5, longContext: true), + "openai/gpt-6-sol": openAI(input: 2, cachedInput: 0.2, output: 10, cacheWrite: 2.5, longContext: true), + "openai/gpt-6-luna": openAI(input: 0.1, cachedInput: 0.01, output: 0.5, cacheWrite: 0.125,longContext: true), + // Promotional standard rate confirmed through at least 2026-11-21. + "openai/gpt-5.6-sol": openAI(input: 4, cachedInput: 0.4, output: 20, cacheWrite: 5, longContext: true), "openai/gpt-5.6-terra": openAI(input: 2, cachedInput: 0.2, output: 12, cacheWrite: 2.5, longContext: true), "openai/gpt-5.6-luna": openAI(input: 0.2, cachedInput: 0.02, output: 1.2, cacheWrite: 0.25, longContext: true), "openai/gpt-5.5": openAI(input: 5, cachedInput: 0.5, output: 30, longContext: true), @@ -106,13 +113,16 @@ public enum Pricing { // The bare selector tokens ("opus", "sonnet", "haiku") are what Claude Code writes // when the user picks a family rather than a version, so they mean *the current* // model of that family, not the one that was current when this table was written. - for alias in ["opus-5", "claude-opus-5", "opus"] { m[alias] = "anthropic/claude-opus-5" } + for alias in ["opus-5.5", "claude-opus-5-5", "opus"] { m[alias] = "anthropic/claude-opus-5-5" } + for alias in ["opus-5", "claude-opus-5"] { m[alias] = "anthropic/claude-opus-5" } for alias in ["opus-4.8", "claude-opus-4-8"] { m[alias] = "anthropic/claude-opus-4-8" } for alias in ["opus-4.7", "claude-opus-4-7"] { m[alias] = "anthropic/claude-opus-4-7" } for alias in ["opus-4.6", "claude-opus-4-6"] { m[alias] = "anthropic/claude-opus-4-6" } for alias in ["opus-4.5", "claude-opus-4-5", "claude-opus-4-5-20251101"] { m[alias] = "anthropic/claude-opus-4-5" } for alias in ["opus-4.1", "claude-opus-4-1", "claude-opus-4-1-20250805"] { m[alias] = "anthropic/claude-opus-4-1" } + for alias in ["fable-5.1", "claude-fable-5-1", "fable"] { m[alias] = "anthropic/claude-fable-5-1" } for alias in ["fable-5", "claude-fable-5"] { m[alias] = "anthropic/claude-fable-5" } + for alias in ["mythos-5.1", "claude-mythos-5-1", "mythos"] { m[alias] = "anthropic/claude-mythos-5-1" } for alias in ["mythos-5", "claude-mythos-5"] { m[alias] = "anthropic/claude-mythos-5" } for alias in ["sonnet-5", "claude-sonnet-5", "sonnet"] { m[alias] = "anthropic/claude-sonnet-5" } for alias in ["sonnet-4.6", "claude-sonnet-4-6"] { m[alias] = "anthropic/claude-sonnet-4-6" } @@ -160,79 +170,57 @@ public enum Pricing { return m }() - /// Returns the canonical pricing key for a raw model string. - public static func normalize(model: String) -> String { + private static func exactCanonicalKey(model: String) -> String? { let lower = model.lowercased() - - // Direct canonical key match if priceTable[lower] != nil { return lower } - - // Exact alias lookup. Provider/router prefixes are allowed only when the final - // path component is a complete known model ID; substring matching made unrelated - // route names ("sonnet-proxy/gpt-…") silently select the wrong provider and price. if let canonical = aliasMap[lower] { return canonical } - if let modelID = lower.split(separator: "/").last, - let canonical = aliasMap[String(modelID)] { - return canonical - } - - // Family keyword fallback (ordered most-specific first). - // - // Each family resolves its version before falling back, and an unrecognized - // version resolves to the *newest* member rather than a pinned one. A bare - // `contains("opus")` used to funnel every Opus into 4.8, so `claude-opus-5` and - // every dated variant of it was silently renamed and merged into the 4.8 row — - // wrong name, wrong grouping, and wrong cost the moment the two tiers diverge. - if lower.contains("opus") { - if lower.contains("4-8") || lower.contains("4.8") { return "anthropic/claude-opus-4-8" } - if lower.contains("4-7") || lower.contains("4.7") { return "anthropic/claude-opus-4-7" } - if lower.contains("4-6") || lower.contains("4.6") { return "anthropic/claude-opus-4-6" } - if lower.contains("4-5") || lower.contains("4.5") { return "anthropic/claude-opus-4-5" } - if lower.contains("4-1") || lower.contains("4.1") { return "anthropic/claude-opus-4-1" } - return "anthropic/claude-opus-5" + let modelID = String(lower.split(separator: "/").last ?? Substring(lower)) + if let canonical = aliasMap[modelID] { return canonical } + // Claude's dated IDs append YYYYMMDD to an otherwise exact model ID. Match + // only that suffix; `claude-opus-5-6` must not inherit Opus 5's exact price. + for key in priceTable.keys where key.hasPrefix("anthropic/") { + let bare = String(key.dropFirst("anthropic/".count)) + guard modelID.hasPrefix(bare + "-") else { continue } + let suffix = modelID.dropFirst(bare.count + 1) + if suffix.count == 8 && suffix.allSatisfy(\.isNumber) { return key } } - if lower.contains("fable") { return "anthropic/claude-fable-5" } - if lower.contains("mythos") { return "anthropic/claude-mythos-5" } - if lower.contains("sonnet") { - if lower.contains("4-6") || lower.contains("4.6") { return "anthropic/claude-sonnet-4-6" } - return "anthropic/claude-sonnet-5" - } - if lower.contains("haiku") { return "anthropic/claude-haiku-4-5" } - if lower.contains("deepseek-v4-flash") { return "deepseek/deepseek-v4-flash" } - if lower.contains("deepseek-v4-pro") { return "deepseek/deepseek-v4-pro" } - if lower.contains("deepseek") { return "deepseek/deepseek-v4-flash" } - if lower.contains("mimo-v2.5-pro") { return "mimo/mimo-v2.5-pro" } - if lower.contains("mimo") { return "mimo/mimo-v2.5" } + return nil + } - // Unknown model: keep its own (lowercased) id rather than collapsing every - // unmatched model into one "_fallback" bucket. It still prices at the - // fallback rate (cost() does priceTable[key] ?? fallback) but the real - // name survives for display. - return lower + /// Returns the canonical pricing key for a known model, otherwise its raw ID. + public static func normalize(model: String) -> String { + exactCanonicalKey(model: model) ?? model.lowercased() } /// False for the generic fallback and for observed provider aliases whose public /// price is only a family estimate. Official canonical IDs, snapshots, and provider- /// prefixed forms of those exact IDs remain exact. public static func priceIsExact(model: String) -> Bool { - priceTable[normalize(model: model)]?.isExact ?? false + guard let key = exactCanonicalKey(model: model) else { return false } + return priceTable[key]?.isExact ?? false } // MARK: - Display names /// Short, prefix-free names for the UI (the provider is shown via the colored dot). private static let displayNames: [String: String] = [ + "anthropic/claude-opus-5-5": "Opus 5.5", "anthropic/claude-opus-5": "Opus 5", "anthropic/claude-opus-4-8": "Opus 4.8", "anthropic/claude-opus-4-7": "Opus 4.7", "anthropic/claude-opus-4-6": "Opus 4.6", "anthropic/claude-opus-4-5": "Opus 4.5", "anthropic/claude-opus-4-1": "Opus 4.1", + "anthropic/claude-fable-5-1": "Fable 5.1", "anthropic/claude-fable-5": "Fable 5", + "anthropic/claude-mythos-5-1": "Mythos 5.1", "anthropic/claude-mythos-5": "Mythos 5", "anthropic/claude-sonnet-5": "Sonnet 5", "anthropic/claude-sonnet-4-6": "Sonnet 4.6", "anthropic/claude-haiku-4-5": "Haiku 4.5", + "openai/gpt-6-astra": "GPT-6 Astra", + "openai/gpt-6-sol": "GPT-6 Sol", + "openai/gpt-6-luna": "GPT-6 Luna", "openai/gpt-5.6-sol": "GPT-5.6 Sol", "openai/gpt-5.6-terra": "GPT-5.6 Terra", "openai/gpt-5.6-luna": "GPT-5.6 Luna", @@ -286,13 +274,17 @@ public enum Pricing { return key } - /// First instant when Sonnet 5 exits its launch price and returns to $3/$15. - private static let sonnet5StandardPriceStarts = Date(timeIntervalSince1970: 1_788_220_800) - private static let sonnet5IntroPrice = ModelPrice(input: 2, output: 10, cacheRead: 0.2, cacheWrite5m: 2.5, cacheWrite1h: 4) - private static func price(forCanonicalKey key: String, at timestamp: Date) -> ModelPrice { - if key == "anthropic/claude-sonnet-5", timestamp < sonnet5StandardPriceStarts { - return sonnet5IntroPrice + _ = timestamp + // Keep known-family estimates for unrecognized versions while preserving the + // raw ID (and priceIsExact == false) in the UI. + if priceTable[key] == nil { + let modelID = String(key.lowercased().split(separator: "/").last ?? Substring(key.lowercased())) + if modelID.hasPrefix("claude-opus-") { return priceTable["anthropic/claude-opus-5-5"]! } + if modelID.hasPrefix("claude-fable-") { return priceTable["anthropic/claude-fable-5-1"]! } + if modelID.hasPrefix("claude-mythos-") { return priceTable["anthropic/claude-mythos-5-1"]! } + if modelID.hasPrefix("claude-sonnet-") { return priceTable["anthropic/claude-sonnet-5"]! } + if modelID.hasPrefix("claude-haiku-") { return priceTable["anthropic/claude-haiku-4-5"]! } } return priceTable[key] ?? fallback } diff --git a/Sources/CodingBarCore/Profile.swift b/Sources/CodingBarCore/Profile.swift index 202f014..cac8167 100644 --- a/Sources/CodingBarCore/Profile.swift +++ b/Sources/CodingBarCore/Profile.swift @@ -24,7 +24,7 @@ enum ProfileBuilder { var hourTokens = [Int](repeating: 0, count: 24) // Favorite = most-frequently used model, not most tokens, so one heavy session // doesn't crown a model the user rarely picks. - var modelCounts: [String: Int] = [:] + var modelCounts: [String: (model: String, provider: Provider, count: Int)] = [:] var dayTokens: [Date: Int] = [:] for r in records { @@ -39,7 +39,11 @@ enum ProfileBuilder { activeDaySet.insert(day) dayTokens[day, default: 0] += r.tokens.total hourTokens[cal.component(.hour, from: r.timestamp)] += r.tokens.total - modelCounts[Pricing.normalize(model: r.model), default: 0] += 1 + let model = Pricing.normalize(model: r.model) + let key = r.provider.rawValue + "·" + model + var entry = modelCounts[key] ?? (model: model, provider: r.provider, count: 0) + entry.count += 1 + modelCounts[key] = entry } let peakHour: Int = { @@ -48,10 +52,11 @@ enum ProfileBuilder { }() // Stable on ties: most uses wins, then the lexicographically smaller key. - let favorite = modelCounts.max { - $0.value != $1.value ? $0.value < $1.value : $0.key > $1.key - }?.key ?? "" - let favoriteProvider = favorite.isEmpty ? Provider.claude : Pricing.provider(forCanonicalKey: favorite) + let favoriteEntry = modelCounts.max { + $0.value.count != $1.value.count ? $0.value.count < $1.value.count : $0.key > $1.key + }?.value + let favorite = favoriteEntry?.model ?? "" + let favoriteProvider = favoriteEntry?.provider ?? .claude let (current, longest) = streaks(activeDays: activeDaySet, now: now, cal: cal) let calendar = contributionCalendar(dayTokens: dayTokens, now: now, cal: cal) diff --git a/Sources/CodingBarCore/Scanner.swift b/Sources/CodingBarCore/Scanner.swift index 433fefb..192226e 100644 --- a/Sources/CodingBarCore/Scanner.swift +++ b/Sources/CodingBarCore/Scanner.swift @@ -83,7 +83,9 @@ final class Scanner { /// preserve the 1-hour prompt-cache portion for duration-aware billing. v7: Codex /// records preserve the absolute last-turn input size for long-context pricing. v8: /// Codex cache-write tokens are split from fresh input instead of being discarded. - private static let cacheVersion = 8 + /// v9: Codex prefers per-response usage records and no longer counts a prior + /// session's cumulative counter as the first turn of each rollout file. + private static let cacheVersion = 9 private struct CacheFile: Codable { var version: Int diff --git a/Tests/CodingBarCoreTests/SmokeTests.swift b/Tests/CodingBarCoreTests/SmokeTests.swift index b6f6690..a4b76f6 100644 --- a/Tests/CodingBarCoreTests/SmokeTests.swift +++ b/Tests/CodingBarCoreTests/SmokeTests.swift @@ -134,7 +134,10 @@ final class SmokeTests: XCTestCase { "payload": ["type": "token_count", "info": ["total_token_usage": ["input_tokens": input, "cached_input_tokens": cached, "cache_write_input_tokens": cacheWrite, - "output_tokens": output, "reasoning_output_tokens": 0]]]] + "output_tokens": output, "reasoning_output_tokens": 0], + "last_token_usage": ["input_tokens": input, "cached_input_tokens": cached, + "cache_write_input_tokens": cacheWrite, + "output_tokens": output, "reasoning_output_tokens": 0]]]] } let lines = [ line(["type": "session_meta", "payload": ["cwd": "/tmp/proj"]]), @@ -167,7 +170,11 @@ final class SmokeTests: XCTestCase { "cache_write_input_tokens": 50_000, "output_tokens": 100, "reasoning_output_tokens": 40, ], - "last_token_usage": ["input_tokens": 300_001], + "last_token_usage": [ + "input_tokens": 300_000, "cached_input_tokens": 100_000, + "cache_write_input_tokens": 50_000, + "output_tokens": 100, "reasoning_output_tokens": 40, + ], ]], ] let lines = [ @@ -188,7 +195,78 @@ final class SmokeTests: XCTestCase { XCTAssertEqual(record.tokens.output, 60, "output_tokens already contains reasoning") XCTAssertEqual(record.tokens.reasoning, 40) XCTAssertEqual(record.tokens.total, 300_100) - XCTAssertEqual(record.billingInputTokens, 300_001) + XCTAssertEqual(record.billingInputTokens, 300_000) + } + + func testModernCodexUsageRecordsIgnoreCumulativeSnapshotsAndDeduplicateResponses() throws { + func line(_ object: [String: Any]) throws -> String { + String(data: try JSONSerialization.data(withJSONObject: object), encoding: .utf8)! + } + func usage(_ id: String, _ timestamp: String, input: Int, cached: Int, output: Int) -> [String: Any] { + ["type": "token_usage_record", "timestamp": timestamp, + "payload": ["response_id": id, + "usage": ["input_tokens": input, "cached_input_tokens": cached, + "cache_write_input_tokens": 0, "output_tokens": output, + "reasoning_output_tokens": 0]]] + } + let snapshot: [String: Any] = ["type": "event_msg", "timestamp": "2026-09-27T10:00:01Z", + "payload": ["type": "token_count", "info": [ + "total_token_usage": ["input_tokens": 50_000_000, "cached_input_tokens": 45_000_000, + "output_tokens": 900_000, "reasoning_output_tokens": 0], + "last_token_usage": ["input_tokens": 300, "cached_input_tokens": 200, + "output_tokens": 10, "reasoning_output_tokens": 0]]]] + let lines: [[String: Any]] = [ + ["type": "session_meta", "payload": ["cwd": "/tmp/project"]], + ["type": "turn_context", "payload": ["model": "gpt-6-astra"]], + ["type": "response_item", "payload": ["type": "function_call", "name": "exec_command"]], + usage("resp-1", "2026-09-27T10:00:00Z", input: 300, cached: 200, output: 10), + snapshot, + usage("resp-1", "2026-09-27T10:00:02Z", input: 300, cached: 200, output: 10), + ["type": "turn_context", "payload": ["model": "gpt-6-luna"]], + usage("resp-2", "2026-09-27T10:05:00Z", input: 400, cached: 300, output: 20), + ] + let url = FileManager.default.temporaryDirectory.appendingPathComponent("rollout-\(UUID().uuidString).jsonl") + try lines.map(line).joined(separator: "\n").write(to: url, atomically: true, encoding: .utf8) + defer { try? FileManager.default.removeItem(at: url) } + + let records = CodexScanner.parseFile(url) + XCTAssertEqual(records.count, 2) + XCTAssertEqual(records.map(\.model), ["gpt-6-astra", "gpt-6-luna"]) + XCTAssertEqual(records.map(\.messageId), ["resp-1", "resp-2"]) + XCTAssertEqual(records.map(\.tokens.input), [100, 100]) + XCTAssertEqual(records.map(\.tokens.cacheRead), [200, 300]) + XCTAssertEqual(records.map(\.tokens.output), [10, 20]) + XCTAssertEqual(records.first?.billingInputTokens, 300) + XCTAssertEqual(records.first?.toolNames, ["exec_command"]) + } + + func testLegacyCodexFirstCounterAndResetUseLastTurnUsage() throws { + func line(_ object: [String: Any]) throws -> String { + String(data: try JSONSerialization.data(withJSONObject: object), encoding: .utf8)! + } + func count(_ timestamp: String, total: Int, last: Int) -> [String: Any] { + ["type": "event_msg", "timestamp": timestamp, + "payload": ["type": "token_count", "info": [ + "total_token_usage": ["input_tokens": total, "cached_input_tokens": 0, + "output_tokens": total / 10, "reasoning_output_tokens": 0], + "last_token_usage": ["input_tokens": last, "cached_input_tokens": 0, + "output_tokens": last / 10, "reasoning_output_tokens": 0]]]] + } + let lines: [[String: Any]] = [ + ["type": "turn_context", "payload": ["model": "gpt-6-astra"]], + count("2026-09-27T10:00:00Z", total: 5_000_000, last: 100), + count("2026-09-27T10:00:01Z", total: 5_000_000, last: 100), + count("2026-09-27T10:05:00Z", total: 5_000_200, last: 200), + count("2026-09-27T10:10:00Z", total: 300, last: 50), + ] + let url = FileManager.default.temporaryDirectory.appendingPathComponent("rollout-\(UUID().uuidString).jsonl") + try lines.map(line).joined(separator: "\n").write(to: url, atomically: true, encoding: .utf8) + defer { try? FileManager.default.removeItem(at: url) } + + let records = CodexScanner.parseFile(url) + XCTAssertEqual(records.count, 3) + XCTAssertEqual(records.map(\.tokens.input), [100, 200, 50]) + XCTAssertEqual(records.map(\.tokens.output), [10, 20, 5]) } /// Codex `function_call` items (exec_command, view_image, …) buffered before a @@ -199,7 +277,9 @@ final class SmokeTests: XCTestCase { func tc(ts: String, input: Int, output: Int) -> [String: Any] { ["type": "event_msg", "timestamp": ts, "payload": ["type": "token_count", "info": ["total_token_usage": ["input_tokens": input, "cached_input_tokens": 0, - "output_tokens": output, "reasoning_output_tokens": 0]]]] + "output_tokens": output, "reasoning_output_tokens": 0], + "last_token_usage": ["input_tokens": input, "cached_input_tokens": 0, + "output_tokens": output, "reasoning_output_tokens": 0]]]] } let lines = [ line(["type": "session_meta", "payload": ["cwd": "/tmp/p"]]), @@ -301,6 +381,30 @@ final class SmokeTests: XCTestCase { XCTAssertEqual(parsed.cacheWrite1h, 60) } + func testClaudeStreamingMessageUsesFinalUsageAndContent() throws { + func line(_ object: [String: Any]) throws -> String { + String(data: try JSONSerialization.data(withJSONObject: object), encoding: .utf8)! + } + func assistant(_ output: Int, content: [[String: Any]]) -> [String: Any] { + ["type": "assistant", "timestamp": "2026-09-27T10:00:00Z", "cwd": "/tmp/project", + "message": ["id": "msg-1", "model": "claude-opus-5-5", "content": content, + "usage": ["input_tokens": 2, "cache_read_input_tokens": 200_000, + "cache_creation_input_tokens": 100, "output_tokens": output]]] + } + let lines = [assistant(2, content: []), + assistant(432, content: [["type": "tool_use", "name": "Bash"]])] + let url = FileManager.default.temporaryDirectory.appendingPathComponent("\(UUID().uuidString).jsonl") + try lines.map(line).joined(separator: "\n").write(to: url, atomically: true, encoding: .utf8) + defer { try? FileManager.default.removeItem(at: url) } + + let parsed = ClaudeScanner.parseFile(url) + let result = ClaudeScanner.deduplicate(parsed) + XCTAssertEqual(result.records.count, 1) + XCTAssertEqual(result.records[0].tokens.output, 432) + XCTAssertEqual(result.records[0].toolNames, ["Bash"]) + XCTAssertEqual(result.seenIds, ["msg-1"]) + } + /// A Codex session that switches model mid-stream (`/model`) must attribute each /// turn to the model in effect AT that turn, not freeze on the session's first one /// (the `model == "unknown"` guard used to ignore every later turn_context). @@ -309,7 +413,9 @@ final class SmokeTests: XCTestCase { func tc(ts: String, input: Int, output: Int) -> [String: Any] { ["type": "event_msg", "timestamp": ts, "payload": ["type": "token_count", "info": ["total_token_usage": ["input_tokens": input, "cached_input_tokens": 0, - "output_tokens": output, "reasoning_output_tokens": 0]]]] + "output_tokens": output, "reasoning_output_tokens": 0], + "last_token_usage": ["input_tokens": input, "cached_input_tokens": 0, + "output_tokens": output, "reasoning_output_tokens": 0]]]] } let lines = [ line(["type": "session_meta", "payload": ["cwd": "/tmp/p"]]), @@ -416,6 +522,10 @@ final class SmokeTests: XCTestCase { } func testPriceIsExactFlagsOnlyFallbackModels() { + XCTAssertTrue(Pricing.priceIsExact(model: "claude-opus-5-5")) + XCTAssertTrue(Pricing.priceIsExact(model: "claude-fable-5-1")) + XCTAssertTrue(Pricing.priceIsExact(model: "claude-mythos-5-1")) + XCTAssertTrue(Pricing.priceIsExact(model: "gpt-6-astra")) XCTAssertTrue(Pricing.priceIsExact(model: "claude-opus-4-8")) XCTAssertTrue(Pricing.priceIsExact(model: "claude-sonnet-5")) XCTAssertTrue(Pricing.priceIsExact(model: "gpt-5.6-sol")) @@ -423,9 +533,27 @@ final class SmokeTests: XCTestCase { XCTAssertFalse(Pricing.priceIsExact(model: "gpt-5.5-codex")) // observed alias, family estimate XCTAssertFalse(Pricing.priceIsExact(model: "gpt-5.6-codex")) // unknown model, generic fallback XCTAssertFalse(Pricing.priceIsExact(model: "totally-unknown-model")) + XCTAssertEqual(Pricing.cost(model: "sonnet-proxy/unknown-model", + tokens: TokenBreakdown(input: 1_000_000), at: Date()), 3, + "a router name must not select a Claude family estimate") + } + + func testModelIdentityIncludesLogProviderForRoutedModels() { + let tokens = TokenBreakdown(input: 100) + let claude = ModelStat(model: "openai/gpt-6-luna", provider: .claude, tokens: tokens, cost: 0.1) + let codex = ModelStat(model: "openai/gpt-6-luna", provider: .codex, tokens: tokens, cost: 0.1) + XCTAssertNotEqual(claude.id, codex.id) + + let now = Date(timeIntervalSince1970: 1_790_500_000) + let record = RawRecord(provider: .claude, model: "gpt-6-luna", timestamp: now, + cwd: "/tmp/project", tokens: tokens, toolName: nil, toolNames: [], + messageId: "routed", sessionKey: "claude-session", hasInterrupt: false) + let profile = ProfileBuilder.build(from: [record], now: now) + XCTAssertEqual(profile.favoriteModel, "openai/gpt-6-luna") + XCTAssertEqual(profile.favoriteModelProvider, .claude) } - func testPricingUsesCacheDurationAndSonnetFiveEffectiveDates() { + func testPricingUsesCacheDurationAndSonnetFivePermanentPrice() { let millionTokens = TokenBreakdown(input: 1_000_000, output: 1_000_000, cacheRead: 1_000_000, cacheWrite: 1_000_000) let july = Date(timeIntervalSince1970: 1_783_555_200) // 2026-07-01 UTC @@ -439,14 +567,23 @@ final class SmokeTests: XCTestCase { XCTAssertEqual(Pricing.cost(model: "claude-sonnet-5", tokens: millionTokens, at: july, cacheWrite1h: 1_000_000), 16.2, accuracy: 0.000_001) XCTAssertEqual(Pricing.cost(model: "claude-sonnet-5", tokens: millionTokens, - at: september, cacheWrite1h: 1_000_000), 24.3, accuracy: 0.000_001) + at: september, cacheWrite1h: 1_000_000), 16.2, accuracy: 0.000_001) + XCTAssertEqual(Pricing.cost(model: "claude-opus-5-5", tokens: millionTokens, + at: september, cacheWrite1h: 1_000_000), 32.2, accuracy: 0.000_001) + XCTAssertEqual(Pricing.cost(model: "claude-fable-5-1", tokens: millionTokens, + at: september, cacheWrite1h: 1_000_000), 80.25, accuracy: 0.000_001) + XCTAssertEqual(Pricing.cost(model: "claude-mythos-5-1", tokens: millionTokens, + at: september, cacheWrite1h: 1_000_000), 80.25, accuracy: 0.000_001) } func testOpenAIModelPricesAndAliasesMatchCurrentTable() { let date = Date(timeIntervalSince1970: 1_786_233_600) // 2026-08-09 UTC let hundredK = 100_000 let cases: [(raw: String, canonical: String, input: Double, cached: Double, output: Double)] = [ - ("gpt-5.6-sol", "openai/gpt-5.6-sol", 5, 0.5, 30), + ("gpt-6-astra", "openai/gpt-6-astra", 10, 1, 50), + ("gpt-6-sol", "openai/gpt-6-sol", 2, 0.2, 10), + ("gpt-6-luna", "openai/gpt-6-luna", 0.1, 0.01, 0.5), + ("gpt-5.6-sol", "openai/gpt-5.6-sol", 4, 0.4, 20), ("gpt-5.6-terra", "openai/gpt-5.6-terra", 2, 0.2, 12), ("gpt-5.6-luna", "openai/gpt-5.6-luna", 0.2, 0.02, 1.2), ("gpt-5.5", "openai/gpt-5.5", 5, 0.5, 30), @@ -526,9 +663,17 @@ final class SmokeTests: XCTestCase { cacheRead: 100_000, cacheWrite: 100_000) XCTAssertEqual(Pricing.cost(model: "gpt-5.6-sol", tokens: tokens, at: date, - billingInputTokens: 272_000), 4.175, accuracy: 0.000_001) + billingInputTokens: 272_000), 2.94, accuracy: 0.000_001) XCTAssertEqual(Pricing.cost(model: "gpt-5.6-sol", tokens: tokens, at: date, - billingInputTokens: 272_001), 6.85, accuracy: 0.000_001) + billingInputTokens: 272_001), 4.88, accuracy: 0.000_001) + XCTAssertEqual(Pricing.cost(model: "gpt-6-astra", tokens: tokens, at: date, + billingInputTokens: 272_000), 7.35, accuracy: 0.000_001) + XCTAssertEqual(Pricing.cost(model: "gpt-6-astra", tokens: tokens, at: date, + billingInputTokens: 272_001), 12.2, accuracy: 0.000_001) + XCTAssertEqual(Pricing.cost(model: "gpt-6-sol", tokens: tokens, at: date, + billingInputTokens: 272_001), 2.44, accuracy: 0.000_001) + XCTAssertEqual(Pricing.cost(model: "gpt-6-luna", tokens: tokens, at: date, + billingInputTokens: 272_001), 0.122, accuracy: 0.000_001) XCTAssertEqual(Pricing.cost(model: "gpt-5.4-mini", tokens: tokens, at: date, billingInputTokens: 500_000), 0.5325, accuracy: 0.000_001, "models without a long-context surcharge must keep their base price") @@ -565,18 +710,20 @@ final class SmokeTests: XCTestCase { /// `normalize` resolved the Opus family with a bare `contains("opus")` that returned /// 4.8, so every `claude-opus-5` record was renamed and merged into the 4.8 row — - /// 11,616 turns and ~$1,127 hidden on one real machine. The *cost* stayed right only - /// because both tiers are $5/$25, which is why nothing looked broken. Each tier must - /// resolve to itself, and an unknown version to the newest rather than a pinned one. - func testEveryModelTierResolvesToItselfAndUnknownsToTheNewest() { + /// 11,616 turns and ~$1,127 hidden on one real machine. Each known tier must + /// resolve to itself, while unknown versions retain their ID and approximate price. + func testEveryModelTierResolvesToItselfAndUnknownsRemainVisible() { for (raw, expected) in [ + ("claude-opus-5-5", "anthropic/claude-opus-5-5"), ("claude-opus-5", "anthropic/claude-opus-5"), ("claude-opus-4-8", "anthropic/claude-opus-4-8"), ("claude-opus-4-7", "anthropic/claude-opus-4-7"), ("claude-opus-4-6", "anthropic/claude-opus-4-6"), ("claude-opus-4-5-20251101", "anthropic/claude-opus-4-5"), ("claude-opus-4-1-20250805", "anthropic/claude-opus-4-1"), + ("claude-fable-5-1", "anthropic/claude-fable-5-1"), ("claude-fable-5", "anthropic/claude-fable-5"), + ("claude-mythos-5-1", "anthropic/claude-mythos-5-1"), ("claude-mythos-5", "anthropic/claude-mythos-5"), ("claude-sonnet-5", "anthropic/claude-sonnet-5"), ("claude-sonnet-4-6", "anthropic/claude-sonnet-4-6"), @@ -584,14 +731,15 @@ final class SmokeTests: XCTestCase { XCTAssertEqual(Pricing.normalize(model: raw), expected, "\(raw) must keep its own identity") } - // Dated variants and unrecognized versions route through the family fallback. + // Dated variants match a known base; future versions remain distinct. XCTAssertEqual(Pricing.normalize(model: "claude-opus-5-20260315"), "anthropic/claude-opus-5") - XCTAssertEqual(Pricing.normalize(model: "claude-opus-9"), "anthropic/claude-opus-5", - "an unknown Opus must resolve to the newest, not a pinned tier") - XCTAssertEqual(Pricing.normalize(model: "claude-sonnet-9"), "anthropic/claude-sonnet-5") + XCTAssertEqual(Pricing.normalize(model: "claude-opus-9"), "claude-opus-9") + XCTAssertFalse(Pricing.priceIsExact(model: "claude-opus-9")) + XCTAssertEqual(Pricing.normalize(model: "claude-sonnet-9"), "claude-sonnet-9") + XCTAssertFalse(Pricing.priceIsExact(model: "claude-sonnet-9")) // The bare selectors Claude Code writes when you pick a family, not a version. - XCTAssertEqual(Pricing.normalize(model: "opus"), "anthropic/claude-opus-5") + XCTAssertEqual(Pricing.normalize(model: "opus"), "anthropic/claude-opus-5-5") XCTAssertEqual(Pricing.normalize(model: "sonnet"), "anthropic/claude-sonnet-5") } diff --git a/release-notes/v1.1.10.md b/release-notes/v1.1.10.md new file mode 100644 index 0000000..10f13cf --- /dev/null +++ b/release-notes/v1.1.10.md @@ -0,0 +1,6 @@ +- Codex usage now reads per-response token records and avoids counting cumulative usage from earlier rollout files. Older logs retain safe delta accounting. +- Claude streamed messages now use their final token counts, correcting understated output and cost. +- Standard API-equivalent pricing now covers GPT-6 Astra, Sol, and Luna, Claude Opus 5.5, Fable 5.1, and Mythos 5.1. GPT-5.6 Sol uses its current promotional rate, Sonnet 5 keeps its permanent launch price, and unknown model versions remain visibly approximate. +- Model breakdowns use the log's actual app source, so GPT models routed through Claude Code no longer appear under Codex. + +**Full Changelog**: https://github.com/Gnonymous/CodingBar/compare/v1.1.9...v1.1.10