From 8c238178a2ff07286a95c1e84bbffdb340eea00a Mon Sep 17 00:00:00 2001 From: Liang-HZ <36034777+Liang-HZ@users.noreply.github.com> Date: Sat, 10 Oct 2026 15:21:57 +0800 Subject: [PATCH] fix(qoder): session model follows the last turn; sub-agent usage from its transcripts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two session-display defects, both verified against real 1.1.65 sessions: 1. summary.model now tracks the LAST NON-EMPTY model of the transcript. The parser used to pin the first assistant record's model, and the details dialog reads summary.model ahead of the turns, so a mid-session switch kept displaying the opening model next to the new model's context window. Blank `model` strings are skipped rather than stored: an empty one would overwrite the last valid model and block the launch-model fallback. 2. Sub-agent spend now reads each run's OWN transcript (/subagents/agent-.jsonl — the same layout claude uses) instead of the run's segment log. One API response streams as several records sharing `message.id`, each repeating the same counters, so calls are deduped by that id; `qoder_turn_usage` splits the cached prefix out (Qoder's counters are Anthropic-named over OpenAI semantics); each call lands on the assistant turn current at ITS OWN timestamp. The fold runs after the context-window occupancy is read — a sub-agent's context is its own. Details dialog, session totals and the Token board now read one number (the board's facts come from turn usage). 3. FACT_SCHEMA_VERSION 4 -> 5: totals change for unchanged transcripts. The segment-log path is gone: it double-counted the cached prefix, could not be told apart from main-chain calls by `turn_id` in every entrypoint, and Qoder rotates it after 30 days. Tests: model switch + blank-model guard; transcript fold (dedup, cache split, per-call turn placement); context window untouched; no-subagents directory unchanged. --- src-tauri/src/commands/token_usage.rs | 9 +- src-tauri/src/parsers/qoder.rs | 340 +++++++++++++++++++++++++- 2 files changed, 339 insertions(+), 10 deletions(-) diff --git a/src-tauri/src/commands/token_usage.rs b/src-tauri/src/commands/token_usage.rs index 1f411ea2ad..a7bed0a33d 100644 --- a/src-tauri/src/commands/token_usage.rs +++ b/src-tauri/src/commands/token_usage.rs @@ -118,6 +118,13 @@ const TOP_CONVERSATIONS: usize = 8; /// the cache-write column stayed 0. Totals are unchanged; the columns are /// not. /// See `parsers::codex::codex_usage_counters`. +/// * `5` — Qoder sub-agent spend reads the run's own transcripts +/// (`/subagents/agent-*.jsonl`) instead of its segment logs: +/// deduped by `message.id`, the cached prefix split out, each call +/// attributed to the turn current at its own timestamp. Totals change for +/// unchanged transcripts — the old path double-counted the cached prefix, +/// and a rotated log silently dropped the spend. See +/// `parsers::qoder::attribute_subagent_usage`. /// /// Only the accounting stored in `token_usage_turn` counts: the four token /// counters, the duration and the timestamp. A conversation's context WINDOW is @@ -132,7 +139,7 @@ const TOP_CONVERSATIONS: usize = 8; /// counters predate it, since a custom/BYO model has always exposed them and /// the parser reads every session under `~/.qoder/projects`, not just the ones /// codeg launched. -const FACT_SCHEMA_VERSION: &str = "4"; +const FACT_SCHEMA_VERSION: &str = "5"; const FACT_SCHEMA_VERSION_KEY: &str = "token_usage_fact_schema_version"; diff --git a/src-tauri/src/parsers/qoder.rs b/src-tauri/src/parsers/qoder.rs index 763d1d6af8..7069696e33 100644 --- a/src-tauri/src/parsers/qoder.rs +++ b/src-tauri/src/parsers/qoder.rs @@ -8,8 +8,8 @@ use serde_json::Value; use walkdir::WalkDir; use crate::models::{ - AgentType, ContentBlock, ConversationDetail, ConversationSummary, MessageRole, TurnUsage, - UnifiedMessage, + AgentType, ContentBlock, ConversationDetail, ConversationSummary, MessageRole, MessageTurn, + TurnRole, TurnUsage, UnifiedMessage, }; use crate::parsers::claude::{ api_error_text, capture_title_record, extract_assistant_content, extract_usage, @@ -297,10 +297,11 @@ impl ReportedContext { /// by construction. /// /// Same correction `parsers::codex::codex_usage_counters` applies for the same -/// reason. Deliberately NOT a `FACT_SCHEMA_VERSION` bump: the only transcripts -/// that can carry non-zero counters are ones recorded with the launch env this -/// release also introduces, so there are no stored rows computed the old way to -/// rebuild. +/// reason. This is what `FACT_SCHEMA_VERSION` 3 exists for: sessions with real +/// counters are not only the ones codeg launched with the usage launch env — +/// a custom/BYO model has always exposed them, and the parser reads every +/// session under `~/.qoder/projects` — so stored rows need a full rebuild +/// under the new rule. fn qoder_turn_usage(value: &Value) -> Option { let mut usage = extract_usage(value)?; let cached = usage @@ -780,7 +781,16 @@ fn parse_transcript(bytes: &[u8]) -> Transcript { .pointer("/message/model") .and_then(Value::as_str) .map(String::from); - if model.is_none() { + // Last NON-EMPTY model wins: a mid-session switch has to land + // in `summary.model` (the details dialog reads it ahead of the + // turns) and in the context-window fallback below. Pinning the + // FIRST model here left both on the session's opening model — + // a switch to Qwen3.8-Flash still displayed DeepSeek-Flash + // next to the new model's 200K window. Blank strings are + // skipped rather than stored: an empty `model` would both + // overwrite the last valid one and block the launch-model + // fallback. + if entry_model.as_deref().is_some_and(|m| !m.trim().is_empty()) { model.clone_from(&entry_model); } let message_id = value @@ -833,7 +843,7 @@ fn parse_transcript(bytes: &[u8]) -> Transcript { if usage.is_some() { last.usage = usage; } - if entry_model.is_some() { + if entry_model.as_deref().is_some_and(|m| !m.trim().is_empty()) { last.model = entry_model; } } else { @@ -1038,8 +1048,16 @@ fn parse_detail(path: &Path, conversation_id: &str) -> Result Result/subagents/agent-.jsonl` — +/// the same layout Claude Code uses (`parsers::claude::attribute_subagent_usage`) +/// — and it is the unit of truth here for the same reasons: the run's own +/// counters ride its assistant records (Anthropic FIELD NAMES over Qoder's +/// OpenAI SEMANTICS, so [`qoder_turn_usage`] splits the cached prefix out), +/// while the alternative — the run's segment log under +/// `logs/sessions///segments/` — is rotated by Qoder's +/// 30-day `sessionRetention`, and in some entrypoints its sub-agent turns are +/// indistinguishable from main-chain ones by `turn_id`. +/// +/// One API response streams as SEVERAL records sharing one `message.id`, each +/// repeating the same counters, so calls are deduped by that id — counted once +/// exactly as the main transcript does (`ClaudeRecordAccumulator`). +/// +/// Every file in the directory is read exactly once, referenced or not +/// (mirroring claude): a sub-agent that was interrupted, or one launched on a +/// branch that was later rewound away, still spent its tokens. The MAIN chain +/// counts only the current branch; sub-agent transcripts do not participate in +/// that walk, so their spend is counted regardless — a deliberate rule, noted +/// here rather than derived. +/// +/// Each call lands on the assistant turn current when THAT call was recorded +/// (by its own timestamp), not on the run's launching turn: a background +/// sub-agent keeps working while the parent has moved on, and per-call +/// placement keeps the turn-level numbers — which the Token board reads — +/// closer to where the spend actually happened. The board attributes the spend +/// to the parent turn's model; that limitation is claude's too. +/// +/// Runs AFTER the context-window occupancy was read: a sub-agent's context is +/// its own and never occupied the parent's window. +fn attribute_subagent_usage(turns: &mut [MessageTurn], session_path: &Path) { + let dir = session_path.with_extension("").join("subagents"); + let Ok(entries) = fs::read_dir(&dir) else { + return; + }; + let mut transcripts: Vec = entries + .flatten() + .map(|e| e.path()) + .filter(|p| p.extension().is_some_and(|x| x == "jsonl")) + .collect(); + // Directory order is filesystem-defined; sort so attribution is identical + // on every parse of the same session. + transcripts.sort(); + + for transcript in transcripts { + for (recorded_at, usage) in subagent_call_usage(&transcript) { + // The turn that was speaking when this call happened. Falling back + // to the last assistant turn keeps the tokens in the session even + // when the record carries no usable timestamp. + let target = recorded_at + .and_then(|ts| { + turns + .iter() + .rposition(|t| matches!(t.role, TurnRole::Assistant) && t.timestamp <= ts) + }) + .or_else(|| { + turns + .iter() + .rposition(|t| matches!(t.role, TurnRole::Assistant)) + }); + let Some(turn) = target.and_then(|i| turns.get_mut(i)) else { + continue; + }; + turn.usage = Some(match turn.usage.take() { + Some(own) => TurnUsage { + input_tokens: own.input_tokens.saturating_add(usage.input_tokens), + output_tokens: own.output_tokens.saturating_add(usage.output_tokens), + cache_creation_input_tokens: own + .cache_creation_input_tokens + .saturating_add(usage.cache_creation_input_tokens), + cache_read_input_tokens: own + .cache_read_input_tokens + .saturating_add(usage.cache_read_input_tokens), + }, + None => usage, + }); + } + } +} + +/// One sub-agent transcript's calls: `(recorded at, usage)` per API response, +/// deduped by `message.id` with the largest counters winning — the same rule +/// the main transcript applies (`ClaudeRecordAccumulator::claim_assistant_usage`). +/// Local rather than the shared `parsers::claude::parse_subagent_tool_calls`: +/// that one reads usage with `extract_usage` (Anthropic semantics, where the +/// cached counters are disjoint), while Qoder's counters need +/// [`qoder_turn_usage`]'s cache split. +fn subagent_call_usage(path: &Path) -> Vec<(Option>, TurnUsage)> { + let Ok(bytes) = fs::read(path) else { + return Vec::new(); + }; + let mut per_call: HashMap>, TurnUsage)> = HashMap::new(); + for line in bytes.split(|b| *b == b'\n') { + let Ok(value) = serde_json::from_slice::(line) else { + continue; + }; + if value.get("type").and_then(Value::as_str) != Some("assistant") { + continue; + } + let Some(id) = value.pointer("/message/id").and_then(Value::as_str) else { + continue; + }; + let Some(usage) = qoder_turn_usage(&value) else { + continue; + }; + let billable = |u: &TurnUsage| { + u.input_tokens + .saturating_add(u.output_tokens) + .saturating_add(u.cache_creation_input_tokens) + .saturating_add(u.cache_read_input_tokens) + }; + match per_call.get_mut(id) { + Some(slot) => { + if billable(&usage) > billable(&slot.1) { + slot.1 = usage; + } + if slot.0.is_none() { + slot.0 = record_timestamp(&value); + } + } + None => { + per_call.insert(id.to_string(), (record_timestamp(&value), usage)); + } + } + } + per_call.into_values().collect() +} + #[cfg(test)] mod tests { use super::*; @@ -1155,6 +1304,179 @@ mod tests { ); } + // A mid-session model switch must reach the summary: the details dialog + // reads `summary.model` ahead of the turns, so a summary pinned to the + // session's OPENING model contradicted the context gauge (which follows + // the last turn's window — Qwen3.8-Flash's 200K next to DeepSeek-Flash). + #[test] + fn summary_model_follows_a_mid_session_switch() { + let tmp = tempfile::tempdir().unwrap(); + const SWITCHED_ANSWER_LINE: &str = r#"{"type":"assistant","uuid":"b1","timestamp":"2026-08-16T15:46:00.000Z","message":{"id":"chatcmpl-2","type":"message","role":"assistant","model":"qfmodel","stop_reason":"end_turn","content":[{"type":"text","text":"second","citations":null}],"usage":{"input_tokens":10,"cache_creation_input_tokens":0,"cache_read_input_tokens":0,"output_tokens":2}},"parentUuid":"a3","isSidechain":false,"cwd":"/private/tmp/probe","sessionId":"s1","userType":"external","entrypoint":"cli","version":"1.1.45","gitBranch":"main"}"#; + write_session( + tmp.path(), + "s1", + &[ + RUNTIME_CONFIG, + USER_LINE, + THINKING_LINE, + TOOL_USE_LINE, + TOOL_RESULT_LINE, + ANSWER_LINE, + SWITCHED_ANSWER_LINE, + ], + ); + + let summaries = parser_in(tmp.path()).list_conversations().unwrap(); + assert_eq!(summaries[0].model.as_deref(), Some("qfmodel")); + let detail = parser_in(tmp.path()).get_conversation("s1").unwrap(); + assert_eq!(detail.summary.model.as_deref(), Some("qfmodel")); + } + + // A blank `model` on a later record must not blank the summary (or block + // the launch-model fallback): only non-empty strings may overwrite the + // last valid one, matching the "Last NON-EMPTY model wins" contract. + #[test] + fn summary_model_ignores_a_blank_later_model() { + let tmp = tempfile::tempdir().unwrap(); + const BLANK_MODEL_LINE: &str = r#"{"type":"assistant","uuid":"b2","timestamp":"2026-08-16T15:47:00.000Z","message":{"id":"chatcmpl-3","type":"message","role":"assistant","model":"","stop_reason":"end_turn","content":[{"type":"text","text":"third","citations":null}],"usage":{"input_tokens":10,"cache_creation_input_tokens":0,"cache_read_input_tokens":0,"output_tokens":2}},"parentUuid":"a3","isSidechain":false,"cwd":"/private/tmp/probe","sessionId":"s1","userType":"external","entrypoint":"cli","version":"1.1.45","gitBranch":"main"}"#; + write_session(tmp.path(), "s1", &[USER_LINE, ANSWER_LINE, BLANK_MODEL_LINE]); + + let summaries = parser_in(tmp.path()).list_conversations().unwrap(); + assert_eq!(summaries[0].model.as_deref(), Some("qmodel_38max")); + let detail = parser_in(tmp.path()).get_conversation("s1").unwrap(); + assert_eq!(detail.summary.model.as_deref(), Some("qmodel_38max")); + } + + /// Write one sub-agent transcript beside a session, in the layout Qoder + /// (and Claude Code) uses: `/subagents/agent-.jsonl`. + fn write_subagent(tmp: &std::path::Path, session_id: &str, agent_id: &str, lines: &[&str]) { + let dir = tmp + .join("projects") + .join("-private-tmp-probe") + .join(session_id) + .join("subagents"); + std::fs::create_dir_all(&dir).unwrap(); + std::fs::write( + dir.join(format!("agent-{agent_id}.jsonl")), + lines.join("\n") + "\n", + ) + .unwrap(); + } + + /// One sub-agent API response as Qoder writes it: Anthropic field NAMES + /// over OpenAI SEMANTICS — `input_tokens` 35,000 of which 34,000 is the + /// cached prefix, `output_tokens` 300. + const SUB_AGENT_CALL: &str = r#"{"type":"assistant","uuid":"c1","timestamp":"2026-08-16T15:45:35.000Z","isSidechain":true,"agentId":"aExplore-1","parent_tool_use_id":"call_1","message":{"id":"msg-sub-1","type":"message","role":"assistant","model":"auto","stop_reason":"end_turn","content":[{"type":"text","text":"done"}],"usage":{"input_tokens":35000,"cache_creation_input_tokens":0,"cache_read_input_tokens":34000,"output_tokens":300}},"version":"1.1.65"}"#; + + // Sub-agent spend lives in the run's OWN transcript + // (`/subagents/agent-.jsonl`, the layout claude uses): + // one API response streams as several records sharing one `message.id`, + // each repeating the same counters, and Qoder's counters need the cache + // split. This pins the whole fold: dedup, split, and turn placement. + #[test] + fn sub_agent_usage_folds_from_its_transcript_with_cache_split() { + let tmp = tempfile::tempdir().unwrap(); + write_session(tmp.path(), "s1", &[USER_LINE, ANSWER_LINE]); + // The same response as Qoder streams it — several fragments, each + // repeating the identical usage block. + write_subagent( + tmp.path(), + "s1", + "aExplore-1", + &[SUB_AGENT_CALL, SUB_AGENT_CALL], + ); + + let detail = parser_in(tmp.path()).get_conversation("s1").unwrap(); + // The call (15:45:35) lands on the assistant turn current then — + // ANSWER_LINE's, at 15:45:34. + let turn = detail + .turns + .iter() + .rev() + .find(|t| matches!(t.role, crate::models::TurnRole::Assistant)) + .expect("assistant turn"); + let usage = turn.usage.clone().expect("turn usage"); + // 35,000 input of which 34,000 cached → 1,000 fresh in; counted ONCE + // despite two records sharing `msg-sub-1`. + assert_eq!(usage.input_tokens, 5 + 1_000); + assert_eq!(usage.cache_read_input_tokens, 5 + 34_000); + assert_eq!(usage.output_tokens, 2 + 300); + + // The details dialog's total is the sum of the turns, sub-agent spend + // included — one number for the dialog and the Token board. + let stats = detail.session_stats.expect("session stats"); + assert_eq!( + stats.total_tokens, + Some(5 + 2 + 5 + 1_000 + 34_000 + 300) + ); + } + + #[test] + fn sub_agent_calls_land_on_the_turn_current_when_each_happened() { + let tmp = tempfile::tempdir().unwrap(); + const SWITCHED_ANSWER_LINE: &str = r#"{"type":"assistant","uuid":"b1","timestamp":"2026-08-16T15:46:00.000Z","message":{"id":"chatcmpl-2","type":"message","role":"assistant","model":"qfmodel","stop_reason":"end_turn","content":[{"type":"text","text":"second","citations":null}],"usage":{"input_tokens":10,"cache_creation_input_tokens":0,"cache_read_input_tokens":0,"output_tokens":2}},"parentUuid":"a3","isSidechain":false,"cwd":"/private/tmp/probe","sessionId":"s1","userType":"external","entrypoint":"cli","version":"1.1.45","gitBranch":"main"}"#; + const SECOND_CALL: &str = r#"{"type":"assistant","uuid":"c2","timestamp":"2026-08-16T15:46:30.000Z","isSidechain":true,"agentId":"aExplore-1","parent_tool_use_id":"call_1","message":{"id":"msg-sub-2","type":"message","role":"assistant","model":"auto","stop_reason":"end_turn","content":[{"type":"text","text":"done"}],"usage":{"input_tokens":35000,"cache_creation_input_tokens":0,"cache_read_input_tokens":34000,"output_tokens":300}},"version":"1.1.65"}"#; + write_session( + tmp.path(), + "s1", + &[USER_LINE, ANSWER_LINE, SWITCHED_ANSWER_LINE], + ); + // First call right after the first answer; second after the switch. + write_subagent( + tmp.path(), + "s1", + "aExplore-1", + &[SUB_AGENT_CALL, SECOND_CALL], + ); + + let detail = parser_in(tmp.path()).get_conversation("s1").unwrap(); + let assistants: Vec<_> = detail + .turns + .iter() + .filter(|t| matches!(t.role, crate::models::TurnRole::Assistant)) + .collect(); + assert_eq!(assistants.len(), 2, "two answers, two turns"); + let first = assistants[0].usage.clone().expect("first turn usage"); + let second = assistants[1].usage.clone().expect("second turn usage"); + // Each call lands on the turn current at ITS OWN timestamp — a + // background sub-agent keeps working after the parent moves on. + assert_eq!(first.input_tokens, 5 + 1_000); + assert_eq!(first.cache_read_input_tokens, 5 + 34_000); + assert_eq!(first.output_tokens, 2 + 300); + assert_eq!(second.input_tokens, 10 + 1_000); + assert_eq!(second.cache_read_input_tokens, 34_000); + assert_eq!(second.output_tokens, 2 + 300); + } + + #[test] + fn sub_agent_usage_does_not_touch_the_context_window() { + let tmp = tempfile::tempdir().unwrap(); + write_session( + tmp.path(), + "s1", + &[RUNTIME_CONFIG, USER_LINE, ANSWER_LINE], + ); + // Control parse: no `subagents/` directory at all. + let baseline = parser_in(tmp.path()).get_conversation("s1").unwrap(); + let baseline_stats = baseline.session_stats.clone().expect("baseline stats"); + assert_eq!(baseline_stats.total_tokens, Some(5 + 2 + 5)); + + write_subagent(tmp.path(), "s1", "aExplore-1", &[SUB_AGENT_CALL]); + let with_sub = parser_in(tmp.path()).get_conversation("s1").unwrap(); + let stats = with_sub.session_stats.expect("stats"); + // A sub-agent's context is its own: occupancy and window are untouched + // (the fold runs after they were read) while the totals grow. + assert_eq!( + stats.context_window_used_tokens, + baseline_stats.context_window_used_tokens + ); + assert_eq!( + stats.context_window_max_tokens, + baseline_stats.context_window_max_tokens + ); + assert!(stats.total_tokens.unwrap() > baseline_stats.total_tokens.unwrap()); + } + #[test] fn detail_builds_turns_with_tool_pairing() { let tmp = tempfile::tempdir().unwrap();