antcolony
All repositories: gitoria
23.0 KB
#!/usr/bin/env node// tests/fake-claude.mjs — stands in for `claude -p` in tests/e2e.mjs (COLONY_CLAUDE=<this file>), so the// `work` process is tested end to end without spending Claude quota. It behaves like Claude Code 2.1.280// with `--output-format json` + `verbose` (a LIST of messages, the `result` entry last).// It is the WORKER, or — when `--json-schema` is the verdict schema — the CONTROLLER (mission 019).// FAKE_CLAUDE_MODE worker: good (default) | nostructured | badreport | hang | falseclaim | longresult// (falseclaim: the report claims `missing.txt`, which is never written; mission 027 longresult:// a `result` of 124 characters + a test step on `.scratch/` → refused → the fix step)// FAKE_DECIDED worker: the report's `decided` (a JSON list; default [] — mission 027)// FAKE_QUESTIONS worker: the report's `questions` (a JSON list; default [] — mission 022)// FAKE_OPEN worker: the report's `open` (a JSON list; default [] — mission 023)// FAKE_WEEK_FILE `/usage`: the weekly percent (default 41; "none" = the answer has no weekly limit — mission 023)// FAKE_CLAUDE_LOG worker: file with argv, cwd, stdin (the brief), whether COLONY_TOKEN_FILE was visible// FAKE_CONTROLLER_MODE controller: pass (default) | fail | noevidence (says pass, one finding "no evidence")// | nostructured | hang// FAKE_QUESTION_RULINGS controller: comma list of rulings per report question (ask|decided|trivial|internal|packed; default ask)// FAKE_CONTROLLER_LOG controller: the same log for the controller call// Mission 020 (the `run` loop):// `-p /usage` the quota probe: answers like Claude 2.1.280's local /usage command (cost 0) with// `usage_report.rate_limits.limits[kind session].percent` = the number in FAKE_USAGE_FILE// (re-read on every call, so a test can change it while `run` runs) or FAKE_USAGE_PERCENT (5)// FAKE_CLAUDE_MODE=ratelimit worker: ends like a session evicted by the usage limit (rate_limit_event// `rejected` with resetsAt = now + FAKE_RATELIMIT_RESET s (default 3), error result// "You've hit your limit …", no structured output); `--resume <id>` then answers like `good`// FAKE_CLAUDE_SLEEP seconds every worker / controller call waits before answering (default 0)// FAKE_CLAUDE_SLEEP_DIR `<folder name>=<seconds>`: the calls run in that folder wait this long instead (antcolony#28)// FAKE_LOG_DIR one JSON file per call: kind (usage/worker/controller), argv, pid, start/end ms, the// COLONY_PORTS* env, cwd — to prove parallel sessions and their port ranges// Mission 025 (the finalize step): a `--resume` whose stdin carries `colony-finalize:` is the FINALIZE call — it// answers by FAKE_FINALIZE_MODE (good | nostructured | badreport | hang; default = the worker's own mode, so an old// test whose worker had no report gets none again), writes no file, logs kind `finalize` (FAKE_LOG_DIR) and its// argv/stdin to FAKE_FINALIZE_LOG. `hang` now writes the worker's file before it hangs (work was done, then time ran out).// Mission 026:// FAKE_COMPLETE worker / finalize: the report's `complete` (default true; "false" = half done)// FAKE_LEAVE_SERVER comma list of call kinds (worker, finalize, controller) that LEAVE servers running on their port// range (detached, like a crashed test's dev servers): worker → COLONY_PORT_FROM (own session env),// +2 (env COLONY_SESSION=s-foreign-e2e: "another colony session"), +3 (ignores SIGTERM);// finalize → +5; controller → +4. Each started server is appended to FAKE_SERVER_LOG (JSON lines// { kind, role, port, pid }) — the e2e checks what the scheduler stopped and kills the rest itself.// prompt cache (simulated): a session remembers its tool list (`--disallowedTools`, '' when none) from its first call;// a `--resume` with the SAME tool list reads the cached prefix (cache read 20000, write 44, cost// 0.0123), a DIFFERENT one re-writes it (cache read 0, write 20000, cost 0.1107 = 9×) — the// effect seen on the first real finalize (mission 025), modelled, not measured.import { readFileSync, writeFileSync, existsSync, mkdirSync, appendFileSync } from 'node:fs';import { join } from 'node:path';import { spawn } from 'node:child_process';import { connect } from 'node:net';const argv = process.argv.slice(2);const opt = (n) => { const i = argv.indexOf(n); return i >= 0 ? argv[i + 1] : null; };const started = Date.now();const callLog = (kind, extra = {}) => {if (!process.env.FAKE_LOG_DIR) return;mkdirSync(process.env.FAKE_LOG_DIR, { recursive: true });const f = join(process.env.FAKE_LOG_DIR, `${started}-${process.pid}-${kind}.json`);writeFileSync(f, JSON.stringify({ kind, argv, pid: process.pid, start: started, end: extra.end || null, cwd: process.cwd(),ports: process.env.COLONY_PORTS || null, portFrom: process.env.COLONY_PORT_FROM || null, portTo: process.env.COLONY_PORT_TO || null,tokenFileVisible: 'COLONY_TOKEN_FILE' in process.env, stopFileVisible: 'COLONY_STOP_FILE' in process.env, ...extra }));};if (argv[0] === '-p' && argv[1] === '/usage') {// the quota probe: a local command in the real Claude Code (no model call, cost 0)let pct = Number(process.env.FAKE_USAGE_PERCENT || 5);if (process.env.FAKE_USAGE_FILE && existsSync(process.env.FAKE_USAGE_FILE)) pct = Number(readFileSync(process.env.FAKE_USAGE_FILE, 'utf8').trim());// mission 023: the weekly percent (default 41) from FAKE_WEEK_FILE (re-read per call); "none" = no weekly limit in the answerlet week = '41';if (process.env.FAKE_WEEK_FILE && existsSync(process.env.FAKE_WEEK_FILE)) week = readFileSync(process.env.FAKE_WEEK_FILE, 'utf8').trim();const resets = new Date(Date.now() + 3 * 3600e3).toISOString().replace('Z', '000+00:00');const text = `You are currently using your subscription to power your Claude Code usage\n\nCurrent session: ${pct}% used · resets soon` + (week === 'none' ? '' : `\nCurrent week (all models): ${week}% used · resets later`);process.stdout.write(JSON.stringify([{ type: 'system', subtype: 'init', permissionMode: opt('--permission-mode'), session_id: 'usage-probe' },{ type: 'assistant', message: { model: '<synthetic>', content: [{ type: 'text', text }] }, usage_report: { session: { total_cost_usd: 0 },rate_limits: { limits: [{ kind: 'session', group: 'session', percent: pct, resets_at: resets }, ...(week === 'none' ? [] : [{ kind: 'weekly_all', group: 'weekly', percent: Number(week), resets_at: resets }])] } } },{ type: 'result', subtype: 'success', is_error: false, num_turns: 0, total_cost_usd: 0, result: text, local_command: 'usage', session_id: 'usage-probe' },]));callLog('usage', { percent: pct, week, end: Date.now() });process.exit(0);}const stdin = readFileSync(0, 'utf8');// Mission 029: the LIBRARIAN (its schema has "supersedes"): reads the JSON input of the prompt and answers by markers in// the creator's text — `#not` (or the text "test") → not a decision; `#global` → project "all"; `#project:<name>`;// `#topic:<word-with-dashes>`; `#excerpt:"…"` → that quote (verbatim); `#paraphrase` → a quote NOT in the text; a text saying// `i never said "X"` supersedes every current entry whose quote contains X. FAKE_LIBRARIAN_MODE good (default) |// nostructured | badproject (answers a project that is not allowed) | hang; FAKE_LIBRARIAN_LOG = file (argv + stdin) per call.// Mission 034: FAKE_LIBRARIAN_MODE_FILE (re-read per call, beats FAKE_LIBRARIAN_MODE) so a test can switch the mode while// `run` runs. `ratelimit` = the shape Claude Code 2.1.280 gave LIVE on 2026-09-25 (librarian/calls/0muha3gdyxbp-result.json):// rate_limit_event rejected (resetsAt = now + FAKE_RATELIMIT_RESET s, default 3), a synthetic assistant message with// error "rate_limit" + "You've hit your session limit · resets 8:50pm (Europe/Vienna)", a result with subtype "success",// is_error true, api_error_status 429, 0 tokens, $0, no structured output. `ratelimit-noreset` = the same WITHOUT the// rate_limit_event and WITHOUT api_error_status (no reset time known; only the synthetic message + the text say it).if (/"supersedes"/.test(opt('--json-schema') || '')) {let lmode = process.env.FAKE_LIBRARIAN_MODE || 'good';if (process.env.FAKE_LIBRARIAN_MODE_FILE && existsSync(process.env.FAKE_LIBRARIAN_MODE_FILE)) lmode = readFileSync(process.env.FAKE_LIBRARIAN_MODE_FILE, 'utf8').trim() || lmode;const at = stdin.indexOf('## Input (JSON)');const line = stdin.slice(at).split('\n').find(l => l.startsWith('{'));const input = JSON.parse(line);const text = input.creatorText.quotes.join('\n');const resetsAt = Math.floor(Date.now() / 1000) + Number(process.env.FAKE_RATELIMIT_RESET || 3);if (process.env.FAKE_LIBRARIAN_LOG) appendFileSync(process.env.FAKE_LIBRARIAN_LOG, JSON.stringify({ argv, cwd: process.cwd(), text, input, mode: lmode, at: Date.now(), resetsAt: lmode === 'ratelimit' ? resetsAt : null, tokenFileVisible: 'COLONY_TOKEN_FILE' in process.env }) + '\n');callLog('librarian', { mode: lmode, text });if (lmode === 'hang') { setInterval(() => {}, 1000); }else if (lmode === 'ratelimit' || lmode === 'ratelimit-noreset') {const sid = 'fake-librarian-limited';const zero = { input_tokens: 0, output_tokens: 0, cache_read_input_tokens: 0, cache_creation_input_tokens: 0 };const msg = "You've hit your session limit · resets 8:50pm (Europe/Vienna)";const out = [{ type: 'system', subtype: 'init', permissionMode: opt('--permission-mode'), session_id: sid, model: 'claude-sonnet-5' }];if (lmode === 'ratelimit') out.push({ type: 'rate_limit_event', rate_limit_info: { status: 'rejected', resetsAt, rateLimitType: 'five_hour', overageStatus: 'rejected', isUsingOverage: false, unifiedWindows: { five_hour: { utilization: 1, resetsAt }, seven_day: { utilization: 0.2, resetsAt: resetsAt + 86400 } } }, session_id: sid });out.push({ type: 'assistant', message: { model: '<synthetic>', role: 'assistant', stop_reason: 'stop_sequence', usage: zero, content: [{ type: 'text', text: msg }] }, session_id: sid, error: 'rate_limit', is_api_error_message: true });const result = { type: 'result', subtype: 'success', is_error: true, num_turns: 1, duration_ms: 488, duration_api_ms: 0, session_id: sid, total_cost_usd: 0, usage: zero, modelUsage: {}, permission_denials: [], terminal_reason: 'api_error', result: msg };if (lmode === 'ratelimit') result.api_error_status = 429;out.push(result);process.stdout.write(JSON.stringify(out));process.exitCode = 1;}else {const not = /#not\b/.test(text) || /^test$/i.test(text.trim());let project = /#global\b/.test(text) ? 'all' : ((text.match(/#project:(\S+)/) || [])[1] || (input.ticket ? input.ticket.project : 'all'));if (lmode === 'badproject') project = 'no-such-project';const topic = ((text.match(/#topic:([a-z-]+)/) || [])[1] || 'general').replace(/-/g, ' ');// #paraphrase: a quote that is NOT in the text (a model that rephrased) — the scheduler must store the whole textconst quote = /#paraphrase\b/.test(text) ? 'the creator wants it teal' : ((text.match(/#excerpt:"([^"]+)"/) || [])[1] || '');const never = (text.match(/i never said "([^"]+)"/i) || [])[1];const supersedes = never ? input.currentEntries.filter(e => e.quote.includes(never)).map(e => e.id) : [];// #longmeans: a reading of 400 characters on two lines (the scheduler makes it one line and clips it to 300)const means = /#longmeans\b/.test(text) ? 'long reading line one ' + 'x'.repeat(180) + '\nline two ' + 'y'.repeat(200) : 'fake reading of: ' + text.replace(/\s+/g, ' ').slice(0, 60);const answer = { decision: !not, project, topic, quote, means: not ? '' : means, supersedes, why: not ? 'fake: marked #not' : 'fake: a decision' };const result = { type: 'result', subtype: lmode === 'nostructured' ? 'error_max_turns' : 'success', is_error: lmode === 'nostructured', num_turns: 1, duration_ms: 321, session_id: 'fake-librarian',total_cost_usd: 0.0031, usage: { input_tokens: 5, output_tokens: 6, cache_read_input_tokens: 7, cache_creation_input_tokens: 8 }, modelUsage: { ['fake-' + opt('--model')]: { costUSD: 0.0031 } }, permission_denials: [] };if (lmode !== 'nostructured') result.structured_output = answer;process.stdout.write(JSON.stringify([{ type: 'system', subtype: 'init', permissionMode: opt('--permission-mode'), session_id: 'fake-librarian' }, result]));process.exitCode = lmode === 'nostructured' ? 1 : 0;}} else {const controller = /"verdict"/.test(opt('--json-schema') || '');const resuming = opt('--resume');let mode = controller ? (process.env.FAKE_CONTROLLER_MODE || 'pass') : (process.env.FAKE_CLAUDE_MODE || 'good');if (resuming && mode === 'ratelimit') mode = 'good';const finalize = !controller && !!resuming && stdin.includes('colony-finalize:');if (finalize && process.env.FAKE_FINALIZE_MODE) mode = process.env.FAKE_FINALIZE_MODE;const logFile = finalize ? (process.env.FAKE_FINALIZE_LOG || process.env.FAKE_CLAUDE_LOG) : controller ? process.env.FAKE_CONTROLLER_LOG : process.env.FAKE_CLAUDE_LOG;if (logFile) writeFileSync(logFile, JSON.stringify({ argv, cwd: process.cwd(), brief: stdin, tokenFileVisible: 'COLONY_TOKEN_FILE' in process.env }, null, 1));const callSession = controller ? (stdin.match(/`session` = `([^`]+)`/) || [])[1] : (stdin.match(/"session": "([^"]+)"/) || [])[1];const kind = controller ? 'controller' : finalize ? 'finalize' : 'worker';callLog(kind, { mode, resume: resuming, session: callSession || null });let usage = { input_tokens: 11, output_tokens: 22, cache_read_input_tokens: 33, cache_creation_input_tokens: 44 };let cost = controller ? 0.0045 : 0.0123;const pm = opt('--permission-mode');const sid = opt('--session-id') || resuming;const rateEvent = (status, resetsAt) => ({ type: 'rate_limit_event', rate_limit_info: { status, resetsAt, rateLimitType: 'five_hour', unifiedWindows: { five_hour: { utilization: status === 'rejected' ? 1 : 0.05, resetsAt } } } });// mission 025: like Claude Code 2.1.280, a RESUMED session reports the cost of the WHOLE session so far (cumulative// total_cost_usd / modelUsage) — the fake keeps each Claude session's total in FAKE_STATE_DIR (default: .fake-claude-sessions/ in the cwd)const tools = opt('--disallowedTools') || '';const sessionTotal = (sid, add) => {if (controller || !sid) return add;const dir = process.env.FAKE_STATE_DIR || join(process.cwd(), '.fake-claude-sessions');mkdirSync(dir, { recursive: true });const f = join(dir, sid + '.json');const st = resuming && existsSync(f) ? JSON.parse(readFileSync(f, 'utf8')) : { total: 0, tools };// mission 026: the simulated prompt cache — a resume with another tool list misses the cached prefixif (resuming) {const hit = (st.tools || '') === tools;usage = { ...usage, cache_read_input_tokens: hit ? 20000 : 0, cache_creation_input_tokens: hit ? 44 : 20000 };if (!hit) add = Math.round(add * 9 * 1e6) / 1e6;}const total = Math.round((st.total + add) * 1e6) / 1e6;writeFileSync(f, JSON.stringify({ total, tools: st.tools === undefined ? tools : st.tools }));return total;};const answer = (structured, messages = []) => {const limited = mode === 'ratelimit';const total = sessionTotal(sid, cost);const failed = mode === 'nostructured' || limited;const resetsAt = Math.floor(Date.now() / 1000) + Number(process.env.FAKE_RATELIMIT_RESET || 3);const result = { type: 'result', subtype: failed && !limited ? 'error_max_turns' : 'success', is_error: failed,num_turns: 2, duration_ms: 1234, session_id: sid, total_cost_usd: total, usage,modelUsage: { ['fake-' + opt('--model')]: { costUSD: total } }, permission_denials: [] };if (limited) { result.result = "You've hit your limit · resets 8:19pm (Europe/Vienna)"; result.api_error_status = 429; }if (!failed) result.structured_output = structured;process.stdout.write(JSON.stringify([{ type: 'system', subtype: 'init', permissionMode: pm, session_id: sid },{ type: 'system', subtype: 'status', permissionMode: pm },rateEvent(limited ? 'rejected' : 'allowed', resetsAt),...messages,result,]));process.stderr.write('fake claude: ' + kind + ' mode ' + mode + (resuming ? ' (resume ' + resuming + ')' : '') + '\n');process.exitCode = failed ? 1 : 0;callLog(kind + '-end', { mode, resume: resuming, end: Date.now(), limited, session: callSession || null });};// mission 026: servers a session leaves behind (detached; they outlive this fake) — waits until each one listensconst serverCode = (stubborn) => `const s=require('http').createServer((q,r)=>r.end('fake'));s.listen(Number(process.env.FAKE_PORT),'127.0.0.1');${stubborn ? "process.on('SIGTERM',()=>{});" : ''}setInterval(()=>{},1000);`;const listening = (port) => new Promise(res => { const c = connect(port, '127.0.0.1'); c.on('connect', () => { c.destroy(); res(true); }); c.on('error', () => res(false)); });const leaveServers = async () => {const kinds = (process.env.FAKE_LEAVE_SERVER || '').split(',');if (!kinds.includes(kind)) return;const from = Number(process.env.COLONY_PORT_FROM);if (!from) return;// a real session starts its servers seconds after it began; the scheduler counts only processes started more than// 50 ms after the session (clock precision margin) — so the fake waits a little tooawait new Promise(r => setTimeout(r, 300));const plan = kind === 'worker' ? [['own', 0, {}, false], ['foreign', 2, { COLONY_SESSION: 's-foreign-e2e' }, false], ['stubborn', 3, {}, true]]: kind === 'finalize' ? [['own', 5, {}, false]] : [['own', 4, {}, false]];for (const [role, off, env, stubborn] of plan) {const port = from + off;const c = spawn(process.execPath, ['-e', serverCode(stubborn), 'fake-left-server-' + kind + '-' + role], { detached: true, stdio: 'ignore', env: { ...process.env, ...env, FAKE_PORT: String(port) } });c.unref();if (process.env.FAKE_SERVER_LOG) appendFileSync(process.env.FAKE_SERVER_LOG, JSON.stringify({ kind, role, port, pid: c.pid }) + '\n');for (let i = 0; i < 50 && !(await listening(port)); i++) await new Promise(r => setTimeout(r, 100));}};const main = async () => {await leaveServers();if (mode === 'hang') {// mission 025: the work was done, then time ran out (not in a finalize call: that one changes nothing)const session = (stdin.match(/"session": "([^"]+)"/) || [])[1];if (!controller && !finalize && session) writeFileSync('fake-worker-was-here.txt', session + '\n');setInterval(() => {}, 1000);}else if (!controller) {const ticket = (stdin.match(/"ticket": "([^"]+)"/) || [])[1];const session = (stdin.match(/"session": "([^"]+)"/) || [])[1];// "the work": a file in the cwd (= the project's dev folder) — the finalize step (mission 025) writes nothingif (!finalize) writeFileSync('fake-worker-was-here.txt', session + '\n');const report = {ticket, session,result: finalize ? 'The fake worker wrote its file; its report came from the finalize step.' : 'The fake worker wrote its file.',complete: process.env.FAKE_COMPLETE !== 'false',test: ['open the dev folder — `fake-worker-was-here.txt` is there', 'it holds the session id'],done: ['created `fake-worker-was-here.txt`'],verified: [{ claim: 'the file exists', command: 'cat fake-worker-was-here.txt', output: session }],open: process.env.FAKE_OPEN ? JSON.parse(process.env.FAKE_OPEN) : [], issues: [], questions: process.env.FAKE_QUESTIONS ? JSON.parse(process.env.FAKE_QUESTIONS) : [],decided: process.env.FAKE_DECIDED ? JSON.parse(process.env.FAKE_DECIDED) : [], running: [], refused: [],};if (mode === 'longresult') {report.result = 'The fake worker wrote its file and this sentence is deliberately far too long for the creator to read in one glance, really.';report.test = ['open .scratch/dev/fake-worker-was-here.txt', 'it holds the session id'];}if (mode === 'badreport') delete report.verified[0].output;if (mode === 'falseclaim') {report.done = ['created `missing.txt` with the content `hello`'];report.verified = [{ claim: '`missing.txt` exists and says hello', command: 'cat missing.txt', output: 'hello' }];report.issues = [{ project: 'hybriel', subject: 'fake issue of a false report', repro: 'r', observed: 'o', expected: 'e', blocks: false }];}// a transcript like Claude's verbose output: one tool call + its resultanswer(report, [{ type: 'assistant', message: { content: [{ type: 'text', text: 'fake worker: writing the file' }, { type: 'tool_use', id: 't1', name: 'Bash', input: { command: 'cat fake-worker-was-here.txt' } }] } },{ type: 'user', message: { content: [{ type: 'tool_result', tool_use_id: 't1', content: session }] } },]);} else {// the controller: really look at the files the report names (like a real controller would)const ticket = (stdin.match(/`ticket` = `([^`]+)`/) || [])[1];const session = (stdin.match(/`session` = `([^`]+)`/) || [])[1];const fenceStart = stdin.indexOf('## The worker report (report.json)');const reportText = stdin.slice(fenceStart).split('\n').find(l => l.startsWith('{'));const report = JSON.parse(reportText);const findings = report.verified.map(v => {const file = (v.command.match(/^cat (\S+)$/) || [])[1];const there = file && existsSync(file);return { claim: v.claim, about: 'verified', status: there ? 'ok' : 'false', check: file ? `test -f ${file}: ${there ? 'exists' : 'no such file'}` : 'not a cat command' };});let verdict = findings.every(f => f.status === 'ok') ? 'pass' : 'fail';if (mode === 'fail') { verdict = 'fail'; findings.push({ claim: 'fake controller says: fail', about: 'other', status: 'false', check: 'FAKE_CONTROLLER_MODE=fail' }); }if (mode === 'noevidence') { verdict = 'pass'; findings.push({ claim: report.done[0], about: 'done', status: 'no evidence', check: 'nothing in the session shows it' }); }// antcolony#24: a ruling per question of the report; FAKE_QUESTION_RULINGS = comma list (default: ask for all)const rl = (process.env.FAKE_QUESTION_RULINGS || '').split(',');const questions = (report.questions || []).map((_, index) => { const ruling = rl[index] || 'ask'; return ruling === 'ask' || ruling === 'packed' ? { index, ruling, reason: 'fake ruling' } : { index, ruling, reason: 'fake ruling', answer: 'Use the obvious choice.' }; });answer({ ticket, session, verdict, findings, questions, summary: `fake controller (${mode}): ${verdict}, ${findings.length} finding(s)` },[{ type: 'assistant', message: { content: [{ type: 'tool_use', id: 'c1', name: 'Bash', input: { command: 'test -f …' } }] } }]);}};let pause = Number(process.env.FAKE_CLAUDE_SLEEP || 0);const byDir = (process.env.FAKE_CLAUDE_SLEEP_DIR || '').split('=');if (byDir.length === 2 && process.cwd().split('/').pop() === byDir[0]) pause = Number(byDir[1]);if (pause > 0) setTimeout(main, pause * 1000); else main();}
Branches
- mainmain branch