From d65ba4de47829a74ba9bd45df21a0e1d17d5285a Mon Sep 17 00:00:00 2001 From: code-crusher Date: Wed, 30 Sep 2026 08:19:36 +0530 Subject: [PATCH 1/2] perf(harness): faster agent loop for OSS models - Replace the deliberation-heavy prompt sections with a short "Working style" block. - Make file_edit/multi_file_edit tolerant of CRLF, tabs vs spaces and trailing whitespace; show the closest region on a miss. - Stub stale bulky tool results past 40% of the context window, auto-compact at 80%, and warn on repeated identical calls. - Kill the whole process group on command timeout or interrupt, so grandchild processes can no longer hang a tool call. - Add bench/: fixture repos, hidden-test tasks and a runner for measuring harness changes. --- .gitignore | 1 + CHANGELOG.md | 30 +++ bench/README.md | 49 ++++ bench/compare.ts | 37 +++ bench/fixture.ts | 363 ++++++++++++++++++++++++++ bench/run.ts | 351 +++++++++++++++++++++++++ bench/tasks.ts | 284 ++++++++++++++++++++ bench/variants.ts | 34 +++ src/core/agent.ts | 256 ++++++++++++++---- src/prompts/system.ts | 23 +- src/tools/executors/executeCommand.ts | 72 ++++- src/tools/executors/files.ts | 165 +++++++++++- src/tools/types.ts | 2 + test/agent-context.test.ts | 203 ++++++++++++++ test/edit-matching.test.ts | 114 ++++++++ test/execute-command.test.ts | 38 +++ 16 files changed, 1922 insertions(+), 100 deletions(-) create mode 100644 bench/README.md create mode 100644 bench/compare.ts create mode 100644 bench/fixture.ts create mode 100644 bench/run.ts create mode 100644 bench/tasks.ts create mode 100644 bench/variants.ts create mode 100644 test/agent-context.test.ts create mode 100644 test/edit-matching.test.ts create mode 100644 test/execute-command.test.ts diff --git a/.gitignore b/.gitignore index 1ea44a3..dddfe95 100644 --- a/.gitignore +++ b/.gitignore @@ -17,3 +17,4 @@ coverage/ .orb/AGENTS.md .orb/links.json .orbcode/AGENTS.md +bench/results/ diff --git a/CHANGELOG.md b/CHANGELOG.md index 2d26c47..b96ea63 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,36 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added + +- **Whitespace-tolerant edits.** `file_edit` / `multi_file_edit` previously required `old_string` to match byte for byte, so a model that reconstructed indentation from memory (tabs vs spaces), or sent LF text for a CRLF file, got "old_string not found" and had to re-read the file and rewrite the edit — an extra model round trip plus another expensive edit-composition step. Matching now falls back, in order, to the same text with the file's line endings, then to a unique line-by-line match that ignores indentation and trailing whitespace (the replacement is re-indented to the file's own style). Replacement text always follows the file's line endings, so a CRLF file is never left with mixed endings. Ambiguous loose matches are still rejected, and successful loose matches say so in the tool result. +- **Actionable "not found" errors.** A failed match now returns the closest region of the file (up to 7 numbered lines with their exact whitespace), so the model can retry without another read. +- **Stale tool-result pruning.** Once context passes 40% of the model's window, bulky results of `read_file`, `search_files`, `list_files`, `execute_command`, `web_fetch` and `web_search` older than the four most recent tool results are sent as one-line stubs. The stored history and session files are untouched; only the outgoing request shrinks, and the boundary advances in batches of six so the request prefix (and the gateway's prompt cache) stays stable between prunes. +- **Automatic compaction.** When context passes 80% of the window, the conversation is summarized mid-turn and the turn continues from the summary, instead of degrading until the user runs `/compact`. A failed compaction is reported once and never retried within the session. +- **Loop warning.** The third identical tool call with identical output and no file edit in between gets an `[OrbCode]` note appended to its result telling the model that repeating it will not change anything. Previously this rule existed only as prompt text. +- **`bench/` harness benchmark.** Fixture repos, tasks with hidden-test verifiers, and a runner (`node --import tsx bench/run.ts --model --label [--reps N] [--suite core|extended|all] [--variant lean] [--steps]`) that drives the real agent loop in isolation and records steps, tokens, reasoning time, time-to-first-token, streaming and tool time, repeated calls and pass/fail. `bench/compare.ts` compares two result files. Not shipped in the npm package. + +### Changed + +- **Leaner system prompt.** The "Plan before editing", "Investigation efficiency" and "Verifying tool results and avoiding loops" sections asked the model to deliberate before every tool call and to write out a full change plan before editing. They are replaced by a four-line "Working style" block (act directly, locate → edit → check once, batch independent calls, never repeat an identical call more than twice). + +### Fixed + +- **Interrupts and timeouts now actually stop shell commands.** `execute_command` killed only the shell on timeout, so a grandchild process (`find /`, a pipeline stage) kept the output pipe open and the tool call hung until it exited on its own — in one benchmark run for 4.6 hours — and pressing Esc never stopped a running command at all. Commands now run in their own process group; a timeout or user interrupt kills the whole group, and the tool result says why the command stopped. + +### Measured impact + +Before/after on the 9-task benchmark (`bench/`, 2 runs per task per model, median per run; all runs pass unless noted): + +| model | wall | steps | input tokens | pass | +|---|---|---|---|---| +| glm-5.3 | 122s → 81s (−33%) | 6 → 5 | −22% | 17/18 → 18/18 | +| glm-5.3-flash | 82s → 66s (−20%) | 5 → 5 | −12% | 18/18 → 18/18 | +| deepseek-v4.1-flash | 25s → 25s | 5 → 5 | −25% | 18/18 → 18/18 | +| gemini-3.8-flash | 72s → 73s | 11 → 11 | −15% | 17/18 → 17/18 | + +The CRLF/tab-indented edit task shows the tolerant-edit change most clearly: glm-5.3-flash went from 15 steps / 202s to 5 steps / 38s, glm-5.3 from 12 steps / 275s to 6 steps / 53s. + ## [6.8.7] - 2026-09-21 ### Added diff --git a/bench/README.md b/bench/README.md new file mode 100644 index 0000000..014ed1f --- /dev/null +++ b/bench/README.md @@ -0,0 +1,49 @@ +# Harness benchmark + +Measures how the agent loop behaves on small, verifiable coding tasks so harness +changes (system prompt, tool schemas, edit tool, context handling) can be judged +on data instead of feel. + +```sh +# baseline vs. a change: run each arm at least 3 times, ideally in parallel +node --import tsx bench/run.ts --model zai/glm-5.3 --label before --reps 3 +node --import tsx bench/run.ts --model zai/glm-5.3 --label after --reps 3 +node --import tsx bench/compare.ts bench/results/before-*.json bench/results/after-*.json +``` + +Flags: `--suite core|extended|all` (default `core`), `--tasks a,b`, `--steps` +(print each step's tools / input tokens), `--variant ` (system-prompt +override from `variants.ts`, for A/B tests before touching `src/`). + +## What it does + +- Every run gets a fresh throwaway git repo (`fixture.ts`) and the real `Agent` + loop with auto-approve on. `HOME` and the config dir are redirected to a temp + dir, so your sessions, `AGENTS.md` and skills never leak in; the login token is + read first and passed explicitly. Unknown model ids are rejected (they would + otherwise fall back to the default silently). +- Success is decided by tests the agent never saw (`tasks.ts`), plus checks such + as "test files untouched" or "CRLF and tabs preserved". +- Results are saved to `bench/results/` (git-ignored). Real model calls cost + money: the core suite is about $0.01 per run on `glm-5.3-flash` and about + $0.08 on `glm-5.3`. + +## Suites + +| suite | tasks | +|---|---| +| `core` | fix-bugs, rename, add-method, question, trivial-edit, multi-file-feature | +| `extended` | legacy-edit (CRLF + tabs), big-repo-bug (40 generated modules), long-feature (5-part change across layers) | + +## Reading the numbers + +- Wall time is dominated by the gateway (time to first token) and is very noisy: + the same task has taken 13s and 41s with identical code, and single steps have + stalled for minutes. Compare **medians over several reps**, and lean on + step count, input tokens and reasoning time, which are steadier. +- Runs that hit a gateway stall (`ECONNRESET`, the 10 minute timeout) fail for + reasons unrelated to the harness; check `detail` / `timedOut` before treating a + FAIL as a regression. +- Context pruning and auto-compaction only trigger at 40% / 80% of the model's + window (about 93k / 186k tokens), which these tasks do not reach; they are + covered by unit tests in `test/agent-context.test.ts` instead. diff --git a/bench/compare.ts b/bench/compare.ts new file mode 100644 index 0000000..279e953 --- /dev/null +++ b/bench/compare.ts @@ -0,0 +1,37 @@ +/** node --import tsx bench/compare.ts — per-arm medians/sums and per-task medians. */ +import * as fs from "node:fs" +import type { RunMetrics } from "./run.js" + +const load = (f: string) => JSON.parse(fs.readFileSync(f, "utf8")) as { label: string; results: RunMetrics[] } +const median = (xs: number[]) => { + const s = [...xs].sort((a, b) => a - b) + const m = s.length >> 1 + return s.length % 2 ? s[m] : (s[m - 1] + s[m]) / 2 +} +const [a, b] = process.argv.slice(2).map(load) +const metrics: [string, (r: RunMetrics) => number][] = [ + ["wall s", (r) => r.wallMs / 1000], + ["think s", (r) => r.reasoningMs / 1000], + ["ttft s", (r) => r.ttftMs / 1000], + ["stream s", (r) => r.streamMs / 1000], + ["steps", (r) => r.steps], + ["input k", (r) => r.inputTokens / 1000], +] +const pct = (x: number, y: number) => (x === 0 ? "n/a" : `${(((y - x) / x) * 100).toFixed(0)}%`) +console.log(`${"".padEnd(10)} ${a.label.padStart(10)} ${b.label.padStart(10)} change (median per run, all tasks)`) +for (const [name, f] of metrics) { + const x = median(a.results.map(f)) + const y = median(b.results.map(f)) + console.log(`${name.padEnd(10)} ${x.toFixed(1).padStart(10)} ${y.toFixed(1).padStart(10)} ${pct(x, y)}`) +} +const sum = (rs: RunMetrics[], f: (r: RunMetrics) => number) => rs.reduce((t, r) => t + f(r), 0) +console.log(`\npass: ${a.label} ${a.results.filter((r) => r.pass).length}/${a.results.length} ${b.label} ${b.results.filter((r) => r.pass).length}/${b.results.length}`) +console.log(`total wall s: ${(sum(a.results, (r) => r.wallMs) / 1000).toFixed(0)} vs ${(sum(b.results, (r) => r.wallMs) / 1000).toFixed(0)} total think s: ${(sum(a.results, (r) => r.reasoningMs) / 1000).toFixed(0)} vs ${(sum(b.results, (r) => r.reasoningMs) / 1000).toFixed(0)}`) +console.log("\nper task (median wall s / think s / steps):") +for (const id of [...new Set(a.results.map((r) => r.task))]) { + const row = (x: { results: RunMetrics[] }) => { + const rs = x.results.filter((r) => r.task === id) + return `${median(rs.map((r) => r.wallMs / 1000)).toFixed(0).padStart(4)} / ${median(rs.map((r) => r.reasoningMs / 1000)).toFixed(0).padStart(3)} / ${median(rs.map((r) => r.steps)).toFixed(0)}` + } + console.log(`${id.padEnd(20)} ${row(a).padEnd(16)} ${row(b)}`) +} diff --git a/bench/fixture.ts b/bench/fixture.ts new file mode 100644 index 0000000..216ea88 --- /dev/null +++ b/bench/fixture.ts @@ -0,0 +1,363 @@ +import { execFileSync } from "node:child_process" +import * as fs from "node:fs" +import * as path from "node:path" + +/** A small ESM inventory library (no dependencies) that the benchmark tasks edit. */ +export const FIXTURE_FILES: Record = { + "package.json": JSON.stringify({ name: "inventory-demo", version: "1.0.0", type: "module", scripts: { test: "node --test" } }, null, 2) + "\n", + + "README.md": `# inventory-demo + +Tiny inventory library. + +\`\`\`js +import { Inventory } from "./src/inventory.js" +import { formatPrice } from "./src/money.js" + +const inv = new Inventory() +inv.add({ sku: "AB-100", name: "widget", category: "tools", priceCents: 1250, quantity: 4 }) +console.log(formatPrice(inv.total())) // $50.00 +\`\`\` + +Run \`npm test\` to run the tests. \`node src/cli.js\` prints a stock report. +`, + + "src/money.js": `/** Format an integer number of cents as a display price. */ +export function formatPrice(cents) { + const sign = cents < 0 ? "-" : "" + const abs = Math.abs(cents) + const dollars = Math.floor(abs / 100) + const rest = String(abs % 100).padStart(2, "0") + return \`\${sign}$\${dollars}.\${rest}\` +} + +export function parsePrice(text) { + const cleaned = text.replace(/[^0-9.]/g, "") + return Math.round(parseFloat(cleaned) * 100) +} +`, + + "src/pricing.js": `/** Apply a percentage discount (0-100) to a price in cents. */ +export function applyDiscount(cents, pct) { + if (pct < 0 || pct > 100) throw new RangeError("pct must be between 0 and 100") + return Math.round(cents * (1 - pct / 100)) +} + +export function addTax(cents, rate) { + return Math.round(cents * (1 + rate)) +} +`, + + "src/utils/strings.js": `export function titleCase(text) { + return text.replace(/\\b\\w/g, (c) => c.toUpperCase()) +} + +export function padRight(text, width) { + return text.length >= width ? text : text + " ".repeat(width - text.length) +} + +export function padLeft(text, width) { + return text.length >= width ? text : " ".repeat(width - text.length) + text +} +`, + + "src/utils/validate.js": `const SKU_PATTERN = /^[A-Z]{2}-\\d{3}$/ + +export function validateSku(sku) { + return SKU_PATTERN.test(sku) +} + +export function assertPositiveInt(value, label) { + if (!Number.isInteger(value) || value < 0) { + throw new TypeError(\`\${label} must be a non-negative integer\`) + } +} +`, + + "src/inventory.js": `import { assertPositiveInt, validateSku } from "./utils/validate.js" + +export class Inventory { + constructor() { + /** @type {Map} */ + this.items = new Map() + } + + add(item) { + if (!validateSku(item.sku)) throw new Error(\`invalid sku: \${item.sku}\`) + assertPositiveInt(item.priceCents, "priceCents") + assertPositiveInt(item.quantity, "quantity") + const existing = this.items.get(item.sku) + if (existing) { + existing.quantity += item.quantity + } else { + this.items.set(item.sku, { ...item }) + } + } + + remove(sku, quantity) { + const item = this.items.get(sku) + if (!item) throw new Error(\`unknown sku: \${sku}\`) + if (quantity > item.quantity) throw new Error("not enough stock") + item.quantity -= quantity + if (item.quantity === 0) this.items.delete(sku) + } + + get(sku) { + return this.items.get(sku) + } + + list() { + return [...this.items.values()] + } + + /** Total value of all stock, in cents. */ + total() { + let sum = 0 + for (const item of this.items.values()) { + sum += item.priceCents * item.quantity + } + return sum + } + + /** Items whose quantity is strictly below the threshold. */ + lowStock(threshold) { + return this.list().filter((item) => item.quantity < threshold) + } +} +`, + + "src/report.js": `import { formatPrice } from "./money.js" +import { padLeft, padRight, titleCase } from "./utils/strings.js" + +export const LOW_STOCK_THRESHOLD = 5 + +export function renderReport(inventory) { + const lines = [padRight("SKU", 10) + padRight("Name", 16) + padLeft("Qty", 5) + padLeft("Price", 10)] + for (const item of inventory.list()) { + lines.push( + padRight(item.sku, 10) + + padRight(titleCase(item.name), 16) + + padLeft(String(item.quantity), 5) + + padLeft(formatPrice(item.priceCents), 10), + ) + } + lines.push("") + lines.push("Total: " + formatPrice(inventory.total())) + const low = inventory.lowStock(LOW_STOCK_THRESHOLD) + if (low.length > 0) { + lines.push("Low stock: " + low.map((item) => item.sku).join(", ")) + } + return lines.join("\\n") +} +`, + + "src/cli.js": `import { Inventory } from "./inventory.js" +import { renderReport } from "./report.js" + +const inventory = new Inventory() +inventory.add({ sku: "AB-100", name: "widget", category: "tools", priceCents: 1250, quantity: 4 }) +inventory.add({ sku: "CD-200", name: "gadget", category: "tools", priceCents: 999, quantity: 12 }) +inventory.add({ sku: "EF-300", name: "gizmo", category: "toys", priceCents: 500, quantity: 20 }) + +console.log(renderReport(inventory)) +`, + + "src/index.js": `export { Inventory } from "./inventory.js" +export { formatPrice, parsePrice } from "./money.js" +export { applyDiscount, addTax } from "./pricing.js" +export { renderReport } from "./report.js" +`, + + "test/money.test.js": `import assert from "node:assert/strict" +import { test } from "node:test" + +import { formatPrice, parsePrice } from "../src/money.js" + +test("formatPrice formats cents", () => { + assert.equal(formatPrice(1234), "$12.34") + assert.equal(formatPrice(5), "$0.05") + assert.equal(formatPrice(-250), "-$2.50") +}) + +test("parsePrice parses text", () => { + assert.equal(parsePrice("$12.34"), 1234) +}) +`, + + "test/pricing.test.js": `import assert from "node:assert/strict" +import { test } from "node:test" + +import { addTax, applyDiscount } from "../src/pricing.js" + +test("applyDiscount", () => { + assert.equal(applyDiscount(1000, 10), 900) + assert.throws(() => applyDiscount(1000, 101), RangeError) +}) + +test("addTax", () => { + assert.equal(addTax(1000, 0.2), 1200) +}) +`, + + "test/inventory.test.js": `import assert from "node:assert/strict" +import { test } from "node:test" + +import { Inventory } from "../src/inventory.js" + +function sample() { + const inv = new Inventory() + inv.add({ sku: "AB-100", name: "widget", category: "tools", priceCents: 1000, quantity: 3 }) + inv.add({ sku: "CD-200", name: "gadget", category: "toys", priceCents: 250, quantity: 8 }) + return inv +} + +test("add merges quantities", () => { + const inv = sample() + inv.add({ sku: "AB-100", name: "widget", category: "tools", priceCents: 1000, quantity: 2 }) + assert.equal(inv.get("AB-100").quantity, 5) +}) + +test("remove deletes empty items", () => { + const inv = sample() + inv.remove("CD-200", 8) + assert.equal(inv.get("CD-200"), undefined) +}) + +test("total is price times quantity", () => { + assert.equal(sample().total(), 3 * 1000 + 8 * 250) +}) + +test("lowStock is strictly below threshold", () => { + const inv = sample() + assert.deepEqual(inv.lowStock(3).map((i) => i.sku), []) + assert.deepEqual(inv.lowStock(4).map((i) => i.sku), ["AB-100"]) +}) +`, + + "test/report.test.js": `import assert from "node:assert/strict" +import { test } from "node:test" + +import { Inventory } from "../src/inventory.js" +import { renderReport } from "../src/report.js" + +test("report shows total", () => { + const inv = new Inventory() + inv.add({ sku: "AB-100", name: "widget", category: "tools", priceCents: 1000, quantity: 10 }) + assert.match(renderReport(inv), /Total: \\$100\\.00/) +}) +`, +} + +/** Bugs injected for the fix task: total() ignores quantity, lowStock is off by one. */ +function injectBugs(dir: string): void { + const file = path.join(dir, "src/inventory.js") + const src = fs + .readFileSync(file, "utf8") + .replace("sum += item.priceCents * item.quantity", "sum += item.priceCents") + .replace("item.quantity < threshold", "item.quantity <= threshold") + fs.writeFileSync(file, src) +} + +/** A legacy module with CRLF line endings and tab indentation, like many real repos. */ +const LEGACY_ORDERS = [ + "export function orderTotal(order) {", + "\tlet total = 0", + "\tfor (const line of order.lines) {", + "\t\ttotal += line.priceCents * line.qty", + "\t}", + "\tif (order.coupon) {", + "\t\ttotal = total - order.coupon.cents", + "\t}", + "\treturn total", + "}", + "", + "export function describeOrder(order) {", + '\treturn "Order " + order.id + ": " + order.lines.length + " lines"', + "}", + "", +].join("\r\n") + +function addLegacy(dir: string): void { + fs.mkdirSync(path.join(dir, "src/legacy"), { recursive: true }) + fs.writeFileSync(path.join(dir, "src/legacy/orders.js"), LEGACY_ORDERS) + fs.writeFileSync( + path.join(dir, "test/orders.test.js"), + [ + 'import assert from "node:assert/strict"', + 'import { test } from "node:test"', + 'import { describeOrder, orderTotal } from "../src/legacy/orders.js"', + "", + 'test("orderTotal sums lines", () => {', + "\tassert.equal(orderTotal({ lines: [{ priceCents: 100, qty: 2 }] }), 200)", + "})", + "", + 'test("describeOrder counts lines", () => {', + '\tassert.match(describeOrder({ id: 1, lines: [1, 2] }), /2 lines/)', + "})", + "", + ].join("\n"), + ) +} + +/** Number of generated modules in the big fixture, and the one carrying the bug. */ +export const BIG_MODULES = 40 +export const BIG_BUG_MODULE = 27 + +/** + * Many similar generated modules plus one test that exercises all of them, with a + * bug in a single module: finding it means running the tests and tracing the + * failure through a large tree instead of a handful of files. + */ +function addBig(dir: string): void { + fs.mkdirSync(path.join(dir, "src/gen"), { recursive: true }) + const exports: string[] = [] + for (let k = 0; k < BIG_MODULES; k++) { + const sign = k === BIG_BUG_MODULE ? "-" : "+" + const filler = Array.from( + { length: 6 }, + (_, j) => + `/** Helper ${j} for bucket ${k}: clamps a value into the bucket range. */\nfunction clamp${j}(value) {\n\tconst low = ${k * 10 + j}\n\tconst high = low + 1000\n\treturn Math.min(high, Math.max(low, value))\n}\n`, + ).join("\n") + fs.writeFileSync( + path.join(dir, `src/gen/mod${k}.js`), + `// Scoring rules for bucket ${k}.\nconst WEIGHT = ${k + 1}\n\n${filler}\n/** Normalize one raw value for bucket ${k}. */\nexport function normalize${k}(value) {\n\treturn value * 2 ${sign} ${k}\n}\n\n/** Weighted score of a list of raw values for bucket ${k}. */\nexport function score${k}(values) {\n\treturn values.reduce((sum, value) => sum + normalize${k}(value) * WEIGHT, 0)\n}\n\nexport const unused${k} = [clamp0, clamp1, clamp2, clamp3, clamp4, clamp5]\n`, + ) + exports.push(`export { score${k} } from "./mod${k}.js"`) + } + fs.writeFileSync(path.join(dir, "src/gen/index.js"), exports.join("\n") + "\n") + const cases = Array.from({ length: BIG_MODULES }, (_, k) => `\t[${k}, score${k}, ${(k + 1) * (12 + 3 * k)}],`).join("\n") + const names = Array.from({ length: BIG_MODULES }, (_, k) => `score${k}`).join(", ") + fs.writeFileSync( + path.join(dir, "test/gen.test.js"), + `import assert from "node:assert/strict"\nimport { test } from "node:test"\nimport { ${names} } from "../src/gen/index.js"\n\nconst cases = [\n${cases}\n]\n\nfor (const [k, fn, expected] of cases) {\n\ttest(\`bucket \${k} scores [1, 2, 3]\`, () => {\n\t\tassert.equal(fn([1, 2, 3]), expected)\n\t})\n}\n`, + ) +} + +export interface FixtureOptions { + /** Inject the planted inventory bugs. */ + bugs?: boolean + /** Add the CRLF + tab-indented legacy module. */ + legacy?: boolean + /** Add the large generated module tree with one buggy module. */ + big?: boolean +} + +/** Write the fixture into `dir` and commit it so the agent sees a clean git repo. */ +export function createFixture(dir: string, options: FixtureOptions = {}): void { + for (const [rel, content] of Object.entries(FIXTURE_FILES)) { + const abs = path.join(dir, rel) + fs.mkdirSync(path.dirname(abs), { recursive: true }) + fs.writeFileSync(abs, content) + } + if (options.bugs) injectBugs(dir) + if (options.legacy) addLegacy(dir) + if (options.big) addBig(dir) + const git = (...args: string[]) => + execFileSync("git", ["-c", "user.name=bench", "-c", "user.email=bench@example.com", ...args], { + cwd: dir, + stdio: "ignore", + }) + git("init", "-q", "-b", "main") + git("add", "-A") + git("commit", "-q", "-m", "initial") +} diff --git a/bench/run.ts b/bench/run.ts new file mode 100644 index 0000000..d29c109 --- /dev/null +++ b/bench/run.ts @@ -0,0 +1,351 @@ +/** + * OrbCode harness benchmark. + * + * node --import tsx bench/run.ts --model zai/glm-5.3-flash --label baseline [--reps 2] [--tasks fix-bugs,rename] + * + * Runs each task in a fresh fixture repo through the real Agent loop (auto-approve) + * and records steps, tokens, reasoning time, tool usage, waste and pass/fail. + * Results land in bench/results/