Files
Colby McHenry db3b8d2a1e test(agent-eval): report what share of explore's bytes the answer used (CG-9)
The envelope view needed a human to say which files answer the question
(`--answer <glob>`). This reads it off the agent's own final answer and
reports one number per run and per call: bytes returned for files the
answer cited, over all bytes returned. That is the #1500 defect as a
number instead of a hunch.

Attribution has two channels, ranked so the weaker one stays separable:
the answer naming the file (reported alone as the conservative floor),
and the answer citing, in a code span, a symbol only that file DEFINES.
Three guards keep the error from leaning optimistic — the direction a
tuning metric must not lean:

  * Only symbols the file defines. Section headers render `name(kind)`
    for call sites too, and crediting those marked excalidraw's
    dragElements.ts used because the answer named `mutateElement`.
  * A definition beats an import alias of the same name (`variable`),
    or `lib/application.js` gets credit for `require('./utils')`.
  * A name on 3+ returned files identifies none of them.

Bare basenames count as citations (agents write `utils.js:225` in prose)
but only for extensions the envelope shipped, so `res.send` and
`mime.contentType` — the same token shape — do not read as files.

Both the envelope view and this share one parse of the rendered markdown
(parseExploreCall), still not the CG-4 sidecar: the sidecar exists only
on a post-CG-4 build and so cannot measure a baseline arm.
2026-08-05 00:42:59 -05:00

131 lines
6.1 KiB
JavaScript

#!/usr/bin/env node
// Parse the newest Claude Code session log for a project + its subagent logs,
// and report the tool-call breakdown (main + subagents). Works for interactive
// runs (driven via itrun.sh) — Claude Code writes full transcripts to
// ~/.claude/projects/<escaped-cwd>/<session>.jsonl with subagents/ alongside.
import { readFileSync, readdirSync, statSync, existsSync, realpathSync } from 'fs';
import { join } from 'path';
import { homedir } from 'os';
import {
classifySufficiency, formatSufficiency,
collectExploreTexts, computeAllocation, finalAnswerText, formatAllocation,
} from './parse-run.mjs';
const projectArg = process.argv[2];
if (!projectArg) { console.error('usage: parse-session.mjs <project-dir>'); process.exit(1); }
// Claude Code escapes the (real) cwd by replacing every "/" with "-".
const real = realpathSync(projectArg);
const escaped = real.replace(/\//g, '-');
const projDir = join(homedir(), '.claude', 'projects', escaped);
if (!existsSync(projDir)) { console.error('no session logs at', projDir); process.exit(1); }
// Newest top-level session .jsonl
const sessions = readdirSync(projDir)
.filter(f => f.endsWith('.jsonl'))
.map(f => ({ f, m: statSync(join(projDir, f)).mtimeMs }))
.sort((a, b) => b.m - a.m);
if (sessions.length === 0) { console.error('no .jsonl sessions in', projDir); process.exit(1); }
const sessionId = sessions[0].f.replace('.jsonl', '');
function tally(file) {
const counts = {};
for (const line of readFileSync(file, 'utf8').split('\n')) {
if (!line) continue;
let ev; try { ev = JSON.parse(line); } catch { continue; }
const content = ev.message?.content;
if (!Array.isArray(content)) continue;
for (const b of content) {
if (b.type === 'tool_use') counts[b.name] = (counts[b.name] || 0) + 1;
}
}
return counts;
}
// Sum token usage from a transcript. The TUI's "Done (…Xk tokens…)" line only
// covers a subagent's throughput; this works for main-thread runs too and is
// consistent across both paths. `gen` = output, `fresh` = uncached input
// (input + cache_creation), `cached` = cache reads (≈free), `total` = all.
function sumTokens(file) {
const t = { gen: 0, fresh: 0, cached: 0 };
for (const line of readFileSync(file, 'utf8').split('\n')) {
if (!line) continue;
let ev; try { ev = JSON.parse(line); } catch { continue; }
const u = ev.message?.usage;
if (!u) continue;
t.gen += u.output_tokens || 0;
t.fresh += (u.input_tokens || 0) + (u.cache_creation_input_tokens || 0);
t.cached += u.cache_read_input_tokens || 0;
}
return t;
}
const mainCounts = tally(join(projDir, sessionId + '.jsonl'));
// Subagent transcripts live under <session>/subagents/*.jsonl
const subDir = join(projDir, sessionId, 'subagents');
const subCounts = {};
let subAgentFiles = 0;
if (existsSync(subDir)) {
for (const f of readdirSync(subDir).filter(f => f.endsWith('.jsonl'))) {
subAgentFiles++;
const c = tally(join(subDir, f));
for (const [k, v] of Object.entries(c)) subCounts[k] = (subCounts[k] || 0) + v;
}
}
const fmt = (counts) => Object.entries(counts).sort((a, b) => b[1] - a[1])
.map(([k, v]) => ` ${String(v).padStart(3)} ${k}`).join('\n') || ' (none)';
console.log(`session: ${sessionId}`);
console.log(`\nMAIN thread tools:\n${fmt(mainCounts)}`);
console.log(`\nSUBAGENT tools (${subAgentFiles} subagent transcript${subAgentFiles === 1 ? '' : 's'}):\n${fmt(subCounts)}`);
const explore = subCounts['mcp__codegraph__codegraph_explore'] || mainCounts['mcp__codegraph__codegraph_explore'] || 0;
const reads = (subCounts['Read'] || 0) + (mainCounts['Read'] || 0);
const greps = (subCounts['Grep'] || 0) + (mainCounts['Grep'] || 0) + (subCounts['Bash'] || 0) + (mainCounts['Bash'] || 0);
console.log(`\nVERDICT: codegraph_explore used ${explore}x | Read ${reads} | Grep/Bash ${greps}`);
// Token totals (main + subagents), consistent across main-thread and subagent runs.
const tok = { gen: 0, fresh: 0, cached: 0 };
const addTok = (t) => { tok.gen += t.gen; tok.fresh += t.fresh; tok.cached += t.cached; };
addTok(sumTokens(join(projDir, sessionId + '.jsonl')));
if (existsSync(subDir)) {
for (const f of readdirSync(subDir).filter(f => f.endsWith('.jsonl'))) addTok(sumTokens(join(subDir, f)));
}
const k = (n) => (n / 1000).toFixed(1) + 'k';
console.log(`TOKENS: gen ${k(tok.gen)} | fresh-in ${k(tok.fresh)} | cached-in ${k(tok.cached)} | billable≈ ${k(tok.gen + tok.fresh)}`);
// What the agent did after each codegraph_explore (CG-8) — the same classifier
// the headless A/B uses, over the interactive transcript.
//
// A subagent's calls live in their OWN file here (headless stream-json
// interleaves them into one stream instead), so they are stitched back in:
// each `agent-*.meta.json` carries the `toolUseId` of the Task that spawned it,
// which is exactly the `parent_tool_use_id` the classifier keys threads on.
// Without that, a delegated search would score as "the agent moved on".
const parseLines = (file, parentToolUseId) => readFileSync(file, 'utf8').split('\n')
.filter(Boolean)
.map((l) => { try { return JSON.parse(l); } catch { return null; } })
.filter(Boolean)
.map((ev) => (parentToolUseId ? { ...ev, parent_tool_use_id: parentToolUseId } : ev));
const events = parseLines(join(projDir, sessionId + '.jsonl'));
if (existsSync(subDir)) {
for (const f of readdirSync(subDir).filter((f) => f.endsWith('.jsonl'))) {
let parent = null;
const meta = join(subDir, f.replace(/\.jsonl$/, '.meta.json'));
if (existsSync(meta)) { try { parent = JSON.parse(readFileSync(meta, 'utf8')).toolUseId ?? null; } catch { /* unreadable */ } }
events.push(...parseLines(join(subDir, f), parent ?? `subagent:${f}`));
}
}
console.log('');
console.log(formatSufficiency({ sufficiency: classifySufficiency(events) }, ''));
// How much of what explore returned the answer drew on (CG-9). An interactive
// transcript has no `result` event, so the answer is the last main-thread
// assistant text — finalAnswerText already falls back to it.
console.log('');
console.log(formatAllocation(
{ allocation: computeAllocation(collectExploreTexts(events), finalAnswerText(events)) }, ''));