diff --git a/.claude/tasks/0031-bench-report-cli.md b/.claude/tasks/0031-bench-report-cli.md
new file mode 100644
index 0000000..9e1e500
--- /dev/null
+++ b/.claude/tasks/0031-bench-report-cli.md
@@ -0,0 +1,34 @@
+---
+spec: benchmark-harness
+blocked-by: [0030-bench-aggregator-and-reporting]
+---
+
+## What to build
+
+The maintainer-facing command that renders the accumulated sample store as the readable comparison, so the aggregator seam built in slice 0030 has a runnable entry point instead of being callable only from code.
+The command opens the store, drains it, aggregates against the scored suite and bonus definitions, and prints the report — `renderReport(aggregate({ records: readAllSamples(store), suite, bonus }))` — to stdout.
+
+It renders whatever has accumulated so far, annotating incomplete coverage rather than blocking on a complete matrix, exactly as the aggregator already does; the command adds only the argument seam and the live store/stdout boundary.
+It is the reporting counterpart to `bench:run` (slice 0029): `bench/report.ts` stands over `bench/aggregate.ts` in the same shape as `bench/run.ts` stands over `bench/run-loop.ts`.
+
+## Acceptance criteria
+
+- [x] A `bench:report` command reads the accumulated store and prints the rendered report to stdout.
+- [x] The store root defaults to `DEFAULT_STORE_ROOT` (`bench/results`) and is overridable with `--store`, matching `bench:run`.
+- [x] The report is produced solely by driving the `readAllSamples` / `aggregate` / `renderReport` seam and the scored suite / bonus definitions from earlier slices — no aggregation, weighting, or rendering is reimplemented in the command.
+- [x] A partial or empty store renders without error (incomplete coverage annotated, an unrun arm shown as an em dash), inheriting the aggregator's behaviour rather than special-casing it.
+- [x] The argument parser is a pure, unit-tested seam (`--store`, `--help`, plus the `--self-review` / `--no-self-review` variant selector — see Implementation Notes), matching the `parseRunArgs` convention. Because the report boundary is offline, the whole command — not just the parser — is deterministic and unit-tested, rather than validated only by running it.
+
+## Implementation Notes
+
+- **Shape mirrors `bench/run.ts`.** `parseReportArgs` is the pure argument seam; `runReportCommand(argv, deps, out)` opens `createSampleStore(storeRoot)`, drives `renderReport(aggregate({ records: readAllSamples(store), suite, bonus }))`, and prints line-by-line through `out`; `main` and the `import.meta`/`argv[1]` direct-execution guard match `run.ts`. Added the `bench:report` script (`tsx bench/report.ts`) and replaced the flagged follow-up note in `bench/README.md` (plus a new "Reading the results" section).
+
+- **The boundary is offline, so the whole command is unit-tested (deviation from the criterion's framing).** Criterion 5 as written assumed the run-command pattern where the live boundary is "validated by running it." But the report reads only the local sample store — no credentials, host, or Agent SDK — so `runReportCommand` is deterministic and is covered by real unit tests over a temp-dir store (populated, empty, and `--help`), not just the parser. It was still verified by running `npm run bench:report` end-to-end. The unused `deps` parameter is retained only for signature parity with the command family (commented at the parameter).
+
+- **Added `--self-review` / `--no-self-review` — beyond the written `(--store, --help)` seam, deliberately.** `aggregate` needs a `suite` and `bonus`, and `buildScoredSuite`/`buildBonusTasks` are parameterized by `selfReviewPermitted`: when self-review is unavailable the two review tasks render as comment reviews in the scored suite and the approve/request-changes pair moves into the bonus catalog. `bench:run` resolves this by probing the live host (`detectSelfReviewSupport`); the offline report has no host to probe, so it cannot detect it and would otherwise have to hardcode one variant silently. The flag makes the choice explicit, defaulting to `true` (matching the richer self-review-permitted configuration). It is well-scoped: the scored coverage is identical either way, so it only selects the bonus capability catalog — documented in `--help` and the README. Criterion 5's "`--store`, `--help`" wording was updated to record it.
+
+- **`DEFAULT_STORE_ROOT` moved to `store.ts`.** It was defined in `run.ts`; its natural home is the store, and both commands now need it, so it lives in `store.ts` as the single source of truth and is re-exported from `run.ts` to keep that command's existing importers (and `run.test.ts`) resolving it unchanged.
+
+- **Tests, TDD.** `bench/report.test.ts` (8 tests) was authored test-first by the test-writer sub-agent from the public interface alone: the parser defaults/overrides/rejection/help, and the command over populated/empty/help paths. As with the `run-loop` slice, the parser is a small cohesive unit whose logic all landed in one GREEN, so the parser cases past the first are passing characterization guards rather than red-first cycles; the `runReportCommand` cell had a genuine red first. Full bench tier green (104 tests), typecheck clean.
+
+- **Review (three-axis, `/review-uncommitted`): Risk overall Low.** Spec axis: all five criteria met; the only finding was that `--self-review` exceeded the written seam — justified in substance, now recorded here and in the criterion. Standards axis: no hard violations; two judgement calls — the `parseReportArgs` flag loop partly overlaps `parseRunArgs` but has genuinely diverged (adds `--no-` negation), left unextracted as the reviewer advised (a shared parser would be premature); and the unused `deps` parameter, addressed with a parity comment.
diff --git a/bench/README.md b/bench/README.md
index 2d44563..1e7fac9 100644
--- a/bench/README.md
+++ b/bench/README.md
@@ -50,8 +50,7 @@ The raw component breakdown is retained on every sample so the data can be re-we
- `run-loop.ts` — `runCells`, the run loop: it runs one chosen `(arm, task)` cell for a batch of trials (defaulting to five, with a reporting floor of three) by driving `runCell` and the sample store, deciding only how many trials to run and at what trial numbers. Deepening a cell continues numbering past the highest trial it already holds and appends, so a cell's sample size grows across sittings rather than being overwritten. It reimplements no orchestration — provision, run, score, and append stay in `runCell`.
- `run.ts` — the maintainer-facing run-loop command. `parseRunArgs` is the pure, unit-tested argument seam; `runBenchCommand` is the live boundary that resolves host access, resolves the scored suite against the host's self-review support, selects the task, and drives `runCells`. Invoked via `npm run bench:run` (see below).
- `aggregate.ts` — the aggregator: the pure seam that renders the accumulated sample store into a readable comparison. `readAllSamples` drains a store into a flat record list; `aggregate` rolls the records up against the task definitions into a `Report` (headline, per-tier and per-token-component breakdowns, and the bonus table); `renderReport` renders that report as stable text. The headline metric, `costEquivalentTokens`, is computed here at render time by weighting the four retained token components by ADR 0014's pricing ratios (fresh input 1×, cache-write 1.25×, cache-read 0.1×, output 5×), so the stored records can be re-weighted without re-running. It reads whatever samples exist and annotates incomplete coverage — cells below the reporting floor (partial) and unsampled cells (missing) — rather than blocking on a complete matrix. A pure function of the records plus the definitions, unit-tested against synthetic append-only sample stores.
-
-The aggregator has no command wrapper yet; a `bench:report` CLI over it is a natural follow-up, in the same shape as `run.ts` is over `run-loop.ts`.
+- `report.ts` — the maintainer-facing reporting command, the counterpart to `run.ts`. `parseReportArgs` is the pure argument seam; `runReportCommand` opens the store, drains it, aggregates against the scored suite and bonus, and prints `renderReport`. Unlike `run.ts` it has no live boundary — it reads only the local store on disk (no credentials, host, or Agent SDK) — so the whole command is deterministic and unit-tested, not smoke-run. Invoked via `npm run bench:report` (see below).
## Running a cell
@@ -67,6 +66,20 @@ Re-running the same cell deepens it — the new trials append rather than overwr
The command is executed with [`tsx`](https://tsx.is) (a devDependency) so the harness's TypeScript runs directly.
Like the runner smoke tier it drives the live host and the Claude Agent SDK, so a real run needs a configured login, the `gitea-axi` CLI on `PATH`, the Agent SDK installed (`npm install @anthropic-ai/claude-agent-sdk` — an optional peer, declared but not installed by default), and a Claude subscription.
+## Reading the results
+
+The reporting command renders whatever has accumulated in the store into the readable comparison — the cost-equivalent-token headline, coverage annotated against the reporting floor, the per-tier and per-token-component breakdowns, and the bonus table:
+
+```
+npm run bench:report
+```
+
+It reads `bench/results/` by default; pass `--store
` to read a different store, and `--help` for the full flag list.
+Incomplete coverage is annotated rather than hidden, so a half-run matrix still renders (unrun arms show an em dash, not a misleading zero).
+
+Unlike `bench:run` this command is offline — it reads only the local store, never the host — so it needs no login, host, or Agent SDK, and is safe to run at any time.
+The `--self-review` / `--no-self-review` flag only selects the bonus capability catalog (whether the approve/request-changes review pair appears there or in the scored suite); the scored coverage is identical either way, so set it to match the host the samples were run on.
+
## Tests
The harness's deterministic seams are unit-tested in this directory, colocated with their source, and run via:
diff --git a/bench/report.test.ts b/bench/report.test.ts
new file mode 100644
index 0000000..1265762
--- /dev/null
+++ b/bench/report.test.ts
@@ -0,0 +1,177 @@
+import { mkdtempSync, rmSync } from "node:fs";
+import { tmpdir } from "node:os";
+import { join } from "node:path";
+import { afterEach, beforeEach, describe, expect, it } from "vitest";
+import type { CliDeps } from "../src/deps.js";
+import type { ResultRecord } from "./result.js";
+import { createSampleStore, DEFAULT_STORE_ROOT } from "./store.js";
+import { parseReportArgs, runReportCommand } from "./report.js";
+
+/** A minimal passing gitea-axi ResultRecord, overridable per sample. */
+function record(overrides: Partial = {}): ResultRecord {
+ return {
+ arm: "gitea-axi",
+ taskId: "t1",
+ tier: "read",
+ trial: 1,
+ timestamp: "2026-07-16T00:00:00Z",
+ tokens: { freshInput: 0, cacheCreation: 0, cacheRead: 0, output: 0 },
+ turns: 0,
+ durationMs: 0,
+ imputedCostUsd: 0,
+ outcome: { pass: true },
+ ...overrides,
+ };
+}
+
+describe("parseReportArgs", () => {
+ // Behavior: with no arguments the parser resolves the documented defaults —
+ // the store root falls back to the module's DEFAULT_STORE_ROOT constant (the
+ // single source of truth for that default, so we assert against the imported
+ // constant rather than the hardcoded "bench/results" string, checking the
+ // parser wires the module default through on omission), and self-review
+ // defaults to permitted, the richer scored variant (independent literal true).
+ it("resolves the documented defaults when no arguments are given", () => {
+ const result = parseReportArgs([]);
+
+ expect(result.help).toBe(false);
+ if (result.help) return;
+
+ expect(result.storeRoot).toBe(DEFAULT_STORE_ROOT);
+ expect(result.selfReview).toBe(true);
+ });
+
+ // Behavior: the spaced value form of --store overrides the store root the report
+ // reads, and --no-self-review flips the self-review variant off. Expected values
+ // are independent literals: the store root the maintainer supplied and false.
+ it("overrides the store root via spaced --store and disables self-review via --no-self-review", () => {
+ const result = parseReportArgs(["--store", "/tmp/bench-out", "--no-self-review"]);
+
+ expect(result.help).toBe(false);
+ if (result.help) return;
+
+ expect(result.storeRoot).toBe("/tmp/bench-out");
+ expect(result.selfReview).toBe(false);
+ });
+
+ // Behavior: --store also accepts the inline --store= form, and --self-review
+ // states the default explicitly (permitted). Expected values are independent
+ // literals: the inline store root and true.
+ it("accepts the inline --store= form and honors an explicit --self-review", () => {
+ const result = parseReportArgs(["--store=/tmp/x", "--self-review"]);
+
+ expect(result.help).toBe(false);
+ if (result.help) return;
+
+ expect(result.storeRoot).toBe("/tmp/x");
+ expect(result.selfReview).toBe(true);
+ });
+
+ // Behavior: malformed input is rejected with a usage error — an unknown flag, a
+ // bare positional argument (not a --flag), a value flag (--store) missing its
+ // value, and a value handed to the boolean --self-review (which takes none).
+ it("rejects malformed input with a usage error", () => {
+ // Unknown flag.
+ expect(() => parseReportArgs(["--frobnicate", "x"])).toThrow();
+ // Bare positional argument, not a --flag.
+ expect(() => parseReportArgs(["positional"])).toThrow();
+ // Value flag missing its value.
+ expect(() => parseReportArgs(["--store"])).toThrow();
+ // The boolean --self-review takes no value.
+ expect(() => parseReportArgs(["--self-review=yes"])).toThrow();
+ });
+
+ // Behavior: --help and its -h alias short-circuit parsing to a help request,
+ // winning even alongside other arguments.
+ it("short-circuits to a help request for --help and -h, even alongside other args", () => {
+ expect(parseReportArgs(["--help"]).help).toBe(true);
+ expect(parseReportArgs(["-h"]).help).toBe(true);
+ expect(parseReportArgs(["--store", "/tmp/x", "--help"]).help).toBe(true);
+ });
+});
+
+describe("runReportCommand", () => {
+ let root: string;
+
+ beforeEach(() => {
+ root = mkdtempSync(join(tmpdir(), "bench-report-"));
+ });
+
+ afterEach(() => {
+ rmSync(root, { recursive: true, force: true });
+ });
+
+ // Behavior: the offline reporting boundary opens the sample store at --store,
+ // drains it, aggregates against the scored suite and bonus, and prints the
+ // rendered comparison line-by-line through `out`, returning exit code 0. Given
+ // three passing gitea-axi samples on one read task — reaching the reporting
+ // floor of three — the rendered output labels the headline metric
+ // ("cost-equivalent", per the spec / ADR 0014), gives every arm a row (the
+ // four arms are gitea-axi, tea, gitea-mcp, raw-api), and annotates coverage
+ // against the "reporting floor of 3". Assertions are on the exit code and the
+ // presence of content, never on column layout, so they survive a rendering
+ // refactor. The command reads no credentials, host, or SDK, so deps are empty.
+ it("opens the store, aggregates, and prints the rendered comparison with exit code 0", async () => {
+ const store = createSampleStore(root);
+ store.append(record({ arm: "gitea-axi", taskId: "t1", tier: "read", trial: 1 }));
+ store.append(record({ arm: "gitea-axi", taskId: "t1", tier: "read", trial: 2 }));
+ store.append(record({ arm: "gitea-axi", taskId: "t1", tier: "read", trial: 3 }));
+
+ const deps: CliDeps = { env: {}, cwd: "/", globals: {} };
+ const lines: string[] = [];
+ const code = await runReportCommand(["--store", root], deps, (l) => lines.push(l));
+ const output = lines.join("\n");
+
+ // Success exit code.
+ expect(code).toBe(0);
+
+ // The headline metric is labelled.
+ expect(output.toLowerCase()).toContain("cost-equivalent");
+
+ // Every arm gets a row.
+ expect(output).toContain("gitea-axi");
+ expect(output).toContain("tea");
+
+ // Coverage is annotated against the reporting floor of three, which the
+ // three samples reach.
+ expect(output).toContain("reporting floor of 3");
+ });
+
+ // Behavior: an empty (never-written) store renders without error, inheriting
+ // the aggregator's placeholder behavior rather than special-casing it. The
+ // headline still labels the metric and every arm, and marks the unrun arms
+ // with the em-dash placeholder ("—", U+2014) rather than a misleading zero —
+ // the established renderReport behavior pinned by bench/aggregate.test.ts.
+ // Returns exit code 0. The `root` from beforeEach is created empty; nothing is
+ // appended.
+ it("renders an empty store without error, marking unrun arms with the em-dash placeholder", async () => {
+ const deps: CliDeps = { env: {}, cwd: "/", globals: {} };
+ const lines: string[] = [];
+ const code = await runReportCommand(["--store", root], deps, (l) => lines.push(l));
+ const output = lines.join("\n");
+
+ // Success exit code.
+ expect(code).toBe(0);
+
+ // The headline metric is labelled and every arm still appears.
+ expect(output.toLowerCase()).toContain("cost-equivalent");
+ expect(output).toContain("gitea-axi");
+
+ // An unrun arm shows the em-dash placeholder, never a zero.
+ expect(output).toContain("—");
+ });
+
+ // Behavior: --help prints the command's help and returns 0 without reading any
+ // store, so it works before a store exists. No store path is created or
+ // referenced. The help text names the command ("bench:report") — an
+ // independent literal.
+ it("prints help naming the command and returns 0 without reading a store", async () => {
+ const deps: CliDeps = { env: {}, cwd: "/", globals: {} };
+ const lines: string[] = [];
+ const code = await runReportCommand(["--help"], deps, (l) => lines.push(l));
+ const output = lines.join("\n");
+
+ expect(code).toBe(0);
+ expect(output).toContain("bench:report");
+ });
+});
diff --git a/bench/report.ts b/bench/report.ts
new file mode 100644
index 0000000..800955b
--- /dev/null
+++ b/bench/report.ts
@@ -0,0 +1,171 @@
+// The maintainer-facing reporting command: render the accumulated sample store
+// into the readable comparison.
+//
+// This is the reporting counterpart to run.ts. Where the run command spends the
+// token budget on one cell, this command reads whatever samples have accumulated
+// so far and prints the aggregator's comparison — headline, coverage, per-tier
+// and per-token-component breakdowns, and the bonus table.
+//
+// Unlike run.ts, this command has no live boundary: it touches only the local
+// sample store on disk (no credentials, host, or Agent SDK), so the whole command
+// is deterministic and unit-tested. `parseReportArgs` is the pure argument seam.
+
+import { pathToFileURL } from "node:url";
+import type { CliDeps } from "../src/deps.js";
+import { aggregate, readAllSamples, renderReport } from "./aggregate.js";
+import { createSampleStore, DEFAULT_STORE_ROOT } from "./store.js";
+import { buildBonusTasks, buildScoredSuite } from "./task-suite.js";
+
+/** A fully-resolved report configuration. */
+export interface ReportArgs {
+ /** The sample-store root to read; defaults to {@link DEFAULT_STORE_ROOT}. */
+ storeRoot: string;
+ /** Whether to render the suite/bonus variant for a self-review-permitting host. */
+ selfReview: boolean;
+}
+
+/** The parse outcome: a request for help, or a resolved configuration to render. */
+export type ParsedReportArgs = { help: true } | ({ help: false } & ReportArgs);
+
+/** The value-taking flags the command understands. */
+const VALUE_FLAGS = new Set(["store"]);
+
+/** The boolean flags the command understands, each with a `--no-` negation. */
+const BOOLEAN_FLAGS = new Set(["self-review"]);
+
+/** A usage error, surfaced to the maintainer with the offending detail. */
+function usage(detail: string): Error {
+ return new Error(`${detail}\n\nUsage: bench:report [--store ] [--self-review | --no-self-review]`);
+}
+
+/**
+ * Parse the report command's argv into a resolved configuration, applying
+ * defaults ({@link DEFAULT_STORE_ROOT} store, self-review permitted). A report
+ * needs no required selection, so no arguments is a valid invocation. Throws a
+ * usage error on an unknown flag, a bare argument, a value-flag missing its
+ * value, or a value handed to a boolean flag.
+ */
+export function parseReportArgs(argv: string[]): ParsedReportArgs {
+ let storeRoot = DEFAULT_STORE_ROOT;
+ let selfReview = true;
+
+ for (let index = 0; index < argv.length; index += 1) {
+ const token = argv[index] as string;
+ if (token === "--help" || token === "-h") {
+ return { help: true };
+ }
+ if (!token.startsWith("--")) {
+ throw usage(`unexpected argument "${token}"`);
+ }
+
+ const equals = token.indexOf("=");
+ const rawName = equals === -1 ? token.slice(2) : token.slice(2, equals);
+ const inlineValue = equals === -1 ? undefined : token.slice(equals + 1);
+
+ // A `--no-` prefix negates a boolean flag.
+ const negated = rawName.startsWith("no-");
+ const name = negated ? rawName.slice(3) : rawName;
+
+ if (BOOLEAN_FLAGS.has(name)) {
+ if (inlineValue !== undefined) {
+ throw usage(`flag --${rawName} takes no value`);
+ }
+ selfReview = !negated;
+ continue;
+ }
+
+ if (negated || !VALUE_FLAGS.has(name)) {
+ throw usage(`unknown flag "--${rawName}"`);
+ }
+
+ let value = inlineValue;
+ if (value === undefined) {
+ value = argv[index + 1];
+ if (value === undefined || value.startsWith("--")) {
+ throw usage(`flag --${name} needs a value`);
+ }
+ index += 1;
+ }
+ if (name === "store") {
+ storeRoot = value;
+ }
+ }
+
+ return { help: false, storeRoot, selfReview };
+}
+
+/** The help text printed for `--help` / `-h`. */
+const HELP_TEXT = `bench:report — render the accumulated benchmark samples into a comparison.
+
+Reads whatever samples have accumulated in the store and prints the aggregator's
+comparison: the cost-equivalent-token headline, coverage annotated against the
+reporting floor, per-tier and per-token-component breakdowns, and the bonus table.
+Incomplete coverage is annotated rather than hidden, so a half-run matrix still
+renders. This command is offline — it reads only the local store, never the host.
+
+Usage:
+ npm run bench:report -- [options]
+
+Options:
+ --store Sample store root to read (default: ${DEFAULT_STORE_ROOT})
+ --self-review Render the variant for a self-review-permitting host (default)
+ --no-self-review Render the variant for a host that forbids self-review
+ -h, --help Show this help
+
+--self-review only affects the bonus capability catalog (whether the approve /
+request-changes review pair appears there or in the scored suite); the scored
+coverage is identical either way. Set it to match the host the samples were run on.`;
+
+/**
+ * Render the accumulated sample store into the readable comparison. This is the
+ * command's boundary, but — unlike the run command — it is offline: it opens the
+ * local store at `--store`, drains it, aggregates against the scored suite and
+ * bonus definitions (resolved against the `--self-review` variant), and prints the
+ * rendered report through `out`. No orchestration, weighting, or rendering is
+ * reimplemented here; it drives the `readAllSamples` / `aggregate` / `renderReport`
+ * seam. Returns a process exit code.
+ */
+export async function runReportCommand(
+ argv: string[],
+ // Unused: an offline report needs no credentials, cwd, or env. Kept for signature
+ // parity with the command family (runBenchCommand takes the same (argv, deps, out)).
+ _deps: CliDeps,
+ out: (line: string) => void,
+): Promise {
+ const parsed = parseReportArgs(argv);
+ if (parsed.help) {
+ out(HELP_TEXT);
+ return 0;
+ }
+
+ const suiteOptions = { selfReviewPermitted: parsed.selfReview };
+ const store = createSampleStore(parsed.storeRoot);
+ const report = aggregate({
+ records: readAllSamples(store),
+ suite: buildScoredSuite(suiteOptions),
+ bonus: buildBonusTasks(suiteOptions),
+ });
+ out(renderReport(report));
+ return 0;
+}
+
+/** Entry point: render the report and set the process exit code. */
+export async function main(): Promise {
+ const deps: CliDeps = { env: process.env, cwd: process.cwd(), globals: {} };
+ try {
+ process.exitCode = await runReportCommand(
+ process.argv.slice(2),
+ deps,
+ (line) => process.stdout.write(`${line}\n`),
+ );
+ } catch (error) {
+ process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`);
+ process.exitCode = 1;
+ }
+}
+
+// Run only when executed directly (e.g. `tsx bench/report.ts`), not when imported
+// by a test. Under a TypeScript runner argv[1] is this file's own path.
+if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) {
+ await main();
+}
diff --git a/bench/run.ts b/bench/run.ts
index b4980e5..e92913c 100644
--- a/bench/run.ts
+++ b/bench/run.ts
@@ -27,18 +27,19 @@ import { DEFAULT_TRIALS, REPORTING_FLOOR, runCells, type RunCellsResult } from "
import { resolveBenchAccess } from "./seed.js";
import { detectSelfReviewSupport } from "./self-review.js";
import { sdkAgentDriver } from "./sdk-driver.js";
-import { createSampleStore } from "./store.js";
+import { createSampleStore, DEFAULT_STORE_ROOT } from "./store.js";
import { buildScoredSuite } from "./task-suite.js";
+// Re-exported so this command's existing importers keep resolving it from here;
+// its single source of truth is now the store, which owns the default root.
+export { DEFAULT_STORE_ROOT };
+
/** The default turn cap a run is bounded by when not overridden. */
export const DEFAULT_TURN_CAP = 40;
/** The default wall-clock backstop (ms) a run is bounded by when not overridden. */
export const DEFAULT_WALL_CLOCK_MS = 300_000;
-/** Where accumulated samples are stored when `--store` is not given. */
-export const DEFAULT_STORE_ROOT = "bench/results";
-
/** Environment variable naming the tea login the benchmark authenticates through. */
export const LOGIN_ENV = "GITEA_AXI_BENCH_LOGIN";
diff --git a/bench/store.ts b/bench/store.ts
index cfea61a..5e1e084 100644
--- a/bench/store.ts
+++ b/bench/store.ts
@@ -2,6 +2,9 @@ import { appendFileSync, mkdirSync, readdirSync, readFileSync } from "node:fs";
import { dirname, join } from "node:path";
import type { CellKey, ResultRecord } from "./result.js";
+/** Where accumulated samples are stored when no store root is given. */
+export const DEFAULT_STORE_ROOT = "bench/results";
+
/** Run `read`, returning `fallback` when the target does not exist yet. */
function ignoreEnoent(read: () => T, fallback: T): T {
try {
diff --git a/package.json b/package.json
index c6bb2f2..f13f638 100644
--- a/package.json
+++ b/package.json
@@ -37,7 +37,8 @@
"test:pack": "vitest run --config vitest.packaging.config.ts",
"test:bench": "vitest run --config vitest.bench.config.ts",
"test:bench:smoke": "vitest run --config vitest.bench-smoke.config.ts",
- "bench:run": "tsx bench/run.ts"
+ "bench:run": "tsx bench/run.ts",
+ "bench:report": "tsx bench/report.ts"
},
"dependencies": {
"@toon-format/toon": "^2.3.0",