Server (SvelteKit + SQLite Docker container) + per-machine agent that parses Claude Code JSONL transcripts. docker-compose one-command deploy; raw transcript view + search gated behind SHOW_TRANSCRIPTS (hidden by default). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
220 lines
7.8 KiB
TypeScript
220 lines
7.8 KiB
TypeScript
/**
|
||
* toknmtr agent — CLI entry point.
|
||
*
|
||
* Walks TRANSCRIPT_ROOT recursively for *.jsonl transcripts (including subagent
|
||
* transcripts under `/subagents/`, which carry real Task/workflow sub-agent usage —
|
||
* often Sonnet/Haiku — attributed to their parent session; only workflow `journal.jsonl`
|
||
* orchestration files are skipped), parses only the bytes appended since the last
|
||
* run (per-file cursor, see cursor.ts), and pushes the resulting events/tool calls/session
|
||
* metadata to the server in one sweep.
|
||
*
|
||
* Usage:
|
||
* node --experimental-strip-types agent/run.ts [--once] [--backfill] [--interval=<ms>]
|
||
*
|
||
* --once Run a single sweep and exit (default behavior is a continuous loop that
|
||
* sweeps every --interval ms until killed).
|
||
* --backfill Ignore stored cursors and reparse every transcript from byte 0 (cursors
|
||
* are still updated afterward, so later normal runs resume incrementally).
|
||
* Implies a single sweep (the point of a backfill is one full catch-up
|
||
* pass, not a repeating one). Safe to re-run — ingest is idempotent.
|
||
* --interval Sweep interval in ms for continuous mode (default 60000).
|
||
*
|
||
* Env: TOKNMTR_URL, TOKNMTR_TOKEN (see push.ts). host = os.hostname().
|
||
*/
|
||
import { closeSync, openSync, readSync, readdirSync, statSync, type Dirent } from 'node:fs';
|
||
import { hostname } from 'node:os';
|
||
import { basename, join } from 'node:path';
|
||
import { getOffset, save as saveCursors, setOffset } from './cursor.ts';
|
||
import {
|
||
parseTranscript,
|
||
TRANSCRIPT_ROOT,
|
||
type ParsedEvent,
|
||
type SessionMeta,
|
||
type ToolCall
|
||
} from './parse.ts';
|
||
import { pushBatch, pushUsageGauges } from './push.ts';
|
||
import { fetchUsageGauges } from './usage.ts';
|
||
|
||
const HOST = hostname();
|
||
const JOURNAL_NAME = 'journal.jsonl';
|
||
const DEFAULT_INTERVAL_MS = 60_000;
|
||
/** How often the subscription-% gauge scrape runs in continuous mode (it's slow + spawns a
|
||
* real `claude` session, so it must NOT run every sweep — mirrors the widget's 5-min cadence). */
|
||
const USAGE_INTERVAL_MS = 5 * 60_000;
|
||
|
||
interface Args {
|
||
backfill: boolean;
|
||
once: boolean;
|
||
intervalMs: number;
|
||
}
|
||
|
||
function parseArgs(argv: string[]): Args {
|
||
const backfill = argv.includes('--backfill');
|
||
const once = argv.includes('--once');
|
||
const intervalArg = argv.find((a) => a.startsWith('--interval='));
|
||
const parsedInterval = intervalArg ? Number(intervalArg.slice('--interval='.length)) : NaN;
|
||
const intervalMs =
|
||
Number.isFinite(parsedInterval) && parsedInterval > 0 ? parsedInterval : DEFAULT_INTERVAL_MS;
|
||
return { backfill, once, intervalMs };
|
||
}
|
||
|
||
function walk(dir: string, out: string[]): void {
|
||
let entries: Dirent[];
|
||
try {
|
||
entries = readdirSync(dir, { withFileTypes: true });
|
||
} catch {
|
||
return; // root doesn't exist yet (fresh machine, no transcripts) — nothing to do
|
||
}
|
||
for (const entry of entries) {
|
||
const full = join(dir, entry.name);
|
||
if (entry.isDirectory()) walk(full, out);
|
||
else if (entry.isFile() && entry.name.endsWith('.jsonl')) out.push(full);
|
||
}
|
||
}
|
||
|
||
/**
|
||
* All transcript files under TRANSCRIPT_ROOT. Includes subagent transcripts
|
||
* (`.../subagents/agent-<id>.jsonl`) — real Task/workflow sub-agent usage that attributes
|
||
* to the parent session via each line's own `sessionId`. Only workflow `journal.jsonl`
|
||
* files are excluded: they carry orchestration bookkeeping (`type:"started"` lines with no
|
||
* `uuid`), not message events, so they'd yield zero events anyway.
|
||
*/
|
||
function listTranscripts(): string[] {
|
||
const out: string[] = [];
|
||
walk(TRANSCRIPT_ROOT, out);
|
||
return out.filter((f) => basename(f) !== JOURNAL_NAME);
|
||
}
|
||
|
||
interface NewBytes {
|
||
text: string;
|
||
newOffset: number;
|
||
}
|
||
|
||
/**
|
||
* Read the byte range [offset, size) of `path` and trim it to the last complete line
|
||
* (i.e. up to and including the last `\n`), so a file mid-write never has a partial JSON
|
||
* line handed to the parser. The trailing partial line (if any) is left for next time —
|
||
* newOffset stays before it.
|
||
*/
|
||
function readNewLines(path: string, offset: number, size: number): NewBytes {
|
||
if (offset >= size) return { text: '', newOffset: offset };
|
||
const length = size - offset;
|
||
const buf = Buffer.alloc(length);
|
||
const fd = openSync(path, 'r');
|
||
try {
|
||
readSync(fd, buf, 0, length, offset);
|
||
} finally {
|
||
closeSync(fd);
|
||
}
|
||
const lastNewline = buf.lastIndexOf(0x0a); // '\n'
|
||
if (lastNewline === -1) return { text: '', newOffset: offset };
|
||
return {
|
||
text: buf.subarray(0, lastNewline + 1).toString('utf8'),
|
||
newOffset: offset + lastNewline + 1
|
||
};
|
||
}
|
||
|
||
interface CursorUpdate {
|
||
path: string;
|
||
offset: number;
|
||
size: number;
|
||
}
|
||
|
||
async function sweep(backfill: boolean): Promise<void> {
|
||
const files = listTranscripts();
|
||
const events: ParsedEvent[] = [];
|
||
const toolCalls: ToolCall[] = [];
|
||
const sessions: SessionMeta[] = [];
|
||
const cursorUpdates: CursorUpdate[] = [];
|
||
let filesWithNewData = 0;
|
||
|
||
for (const path of files) {
|
||
let size: number;
|
||
try {
|
||
size = statSync(path).size;
|
||
} catch {
|
||
continue; // file disappeared mid-walk — skip
|
||
}
|
||
|
||
const offset = backfill ? 0 : getOffset(path, size);
|
||
const { text, newOffset } = readNewLines(path, offset, size);
|
||
if (!text) continue;
|
||
|
||
const sessionIdFallback = basename(path, '.jsonl');
|
||
const result = parseTranscript(HOST, sessionIdFallback, text);
|
||
events.push(...result.events);
|
||
toolCalls.push(...result.toolCalls);
|
||
if (result.events.length > 0 || result.toolCalls.length > 0) sessions.push(result.session);
|
||
|
||
cursorUpdates.push({ path, offset: newOffset, size });
|
||
filesWithNewData++;
|
||
}
|
||
|
||
const pushResult = await pushBatch({ host: HOST, events, toolCalls, sessions });
|
||
|
||
// Only commit cursors after a successful push, so a failed/unreachable-server push
|
||
// gets retried (nothing is silently lost) on the next sweep. This runs even after a
|
||
// --backfill sweep (which ignored the *old* cursor values when deciding where to start
|
||
// reading) so subsequent normal runs pick up incrementally from here rather than
|
||
// re-walking full history every time.
|
||
for (const u of cursorUpdates) setOffset(u.path, u.offset, u.size);
|
||
if (cursorUpdates.length > 0) saveCursors();
|
||
|
||
console.log(
|
||
`toknmtr agent: scanned ${files.length} transcript(s), ${filesWithNewData} with new data, ` +
|
||
`pushed ${pushResult.received} event(s) in ${pushResult.requests} request(s) ` +
|
||
`(${toolCalls.length} tool call(s), ${sessions.length} session(s)).`
|
||
);
|
||
}
|
||
|
||
/**
|
||
* Best-effort subscription-% gauge scrape + push. Never throws — a missing PTY driver, no
|
||
* `claude` on PATH, or an unreachable server must not break the transcript sweep. Disabled
|
||
* entirely by setting TOKNMTR_NO_USAGE=1 (e.g. on a headless container with no `claude`).
|
||
*/
|
||
async function sweepUsage(): Promise<void> {
|
||
if (process.env.TOKNMTR_NO_USAGE === '1') return;
|
||
try {
|
||
const gauges = await fetchUsageGauges();
|
||
if (!gauges) return;
|
||
await pushUsageGauges(gauges);
|
||
console.log(
|
||
`toknmtr agent: pushed usage gauges (session=${gauges.session_pct ?? '–'}% ` +
|
||
`week=${gauges.week_all_pct ?? '–'}% sonnet=${gauges.week_sonnet_pct ?? '–'}%).`
|
||
);
|
||
} catch (err) {
|
||
console.error('toknmtr agent: usage gauge scrape/push failed (non-fatal):', err);
|
||
}
|
||
}
|
||
|
||
async function main(): Promise<void> {
|
||
const args = parseArgs(process.argv.slice(2));
|
||
|
||
// A backfill is a pure historical-transcript catch-up: skip the (slow, live-only) gauge scrape.
|
||
if (args.backfill) {
|
||
await sweep(true);
|
||
return;
|
||
}
|
||
|
||
if (args.once) {
|
||
await sweep(false);
|
||
await sweepUsage();
|
||
return;
|
||
}
|
||
|
||
// Continuous mode: sweep transcripts every --interval, scrape usage gauges on a slower cadence.
|
||
let lastUsageAt = 0;
|
||
for (;;) {
|
||
await sweep(false);
|
||
if (Date.now() - lastUsageAt >= USAGE_INTERVAL_MS) {
|
||
await sweepUsage();
|
||
lastUsageAt = Date.now();
|
||
}
|
||
await new Promise((resolve) => setTimeout(resolve, args.intervalMs));
|
||
}
|
||
}
|
||
|
||
main().catch((err: unknown) => {
|
||
console.error('toknmtr agent failed:', err);
|
||
process.exitCode = 1;
|
||
});
|