diff --git a/.claude/tasks/0031-bench-report-cli.md b/.claude/tasks/0031-bench-report-cli.md new file mode 100644 index 0000000..9e1e500 --- /dev/null +++ b/.claude/tasks/0031-bench-report-cli.md @@ -0,0 +1,34 @@ +--- +spec: benchmark-harness +blocked-by: [0030-bench-aggregator-and-reporting] +--- + +## What to build + +The maintainer-facing command that renders the accumulated sample store as the readable comparison, so the aggregator seam built in slice 0030 has a runnable entry point instead of being callable only from code. +The command opens the store, drains it, aggregates against the scored suite and bonus definitions, and prints the report — `renderReport(aggregate({ records: readAllSamples(store), suite, bonus }))` — to stdout. + +It renders whatever has accumulated so far, annotating incomplete coverage rather than blocking on a complete matrix, exactly as the aggregator already does; the command adds only the argument seam and the live store/stdout boundary. +It is the reporting counterpart to `bench:run` (slice 0029): `bench/report.ts` stands over `bench/aggregate.ts` in the same shape as `bench/run.ts` stands over `bench/run-loop.ts`. + +## Acceptance criteria + +- [x] A `bench:report` command reads the accumulated store and prints the rendered report to stdout. +- [x] The store root defaults to `DEFAULT_STORE_ROOT` (`bench/results`) and is overridable with `--store`, matching `bench:run`. +- [x] The report is produced solely by driving the `readAllSamples` / `aggregate` / `renderReport` seam and the scored suite / bonus definitions from earlier slices — no aggregation, weighting, or rendering is reimplemented in the command. +- [x] A partial or empty store renders without error (incomplete coverage annotated, an unrun arm shown as an em dash), inheriting the aggregator's behaviour rather than special-casing it. +- [x] The argument parser is a pure, unit-tested seam (`--store`, `--help`, plus the `--self-review` / `--no-self-review` variant selector — see Implementation Notes), matching the `parseRunArgs` convention. Because the report boundary is offline, the whole command — not just the parser — is deterministic and unit-tested, rather than validated only by running it. + +## Implementation Notes + +- **Shape mirrors `bench/run.ts`.** `parseReportArgs` is the pure argument seam; `runReportCommand(argv, deps, out)` opens `createSampleStore(storeRoot)`, drives `renderReport(aggregate({ records: readAllSamples(store), suite, bonus }))`, and prints line-by-line through `out`; `main` and the `import.meta`/`argv[1]` direct-execution guard match `run.ts`. Added the `bench:report` script (`tsx bench/report.ts`) and replaced the flagged follow-up note in `bench/README.md` (plus a new "Reading the results" section). + +- **The boundary is offline, so the whole command is unit-tested (deviation from the criterion's framing).** Criterion 5 as written assumed the run-command pattern where the live boundary is "validated by running it." But the report reads only the local sample store — no credentials, host, or Agent SDK — so `runReportCommand` is deterministic and is covered by real unit tests over a temp-dir store (populated, empty, and `--help`), not just the parser. It was still verified by running `npm run bench:report` end-to-end. The unused `deps` parameter is retained only for signature parity with the command family (commented at the parameter). + +- **Added `--self-review` / `--no-self-review` — beyond the written `(--store, --help)` seam, deliberately.** `aggregate` needs a `suite` and `bonus`, and `buildScoredSuite`/`buildBonusTasks` are parameterized by `selfReviewPermitted`: when self-review is unavailable the two review tasks render as comment reviews in the scored suite and the approve/request-changes pair moves into the bonus catalog. `bench:run` resolves this by probing the live host (`detectSelfReviewSupport`); the offline report has no host to probe, so it cannot detect it and would otherwise have to hardcode one variant silently. The flag makes the choice explicit, defaulting to `true` (matching the richer self-review-permitted configuration). It is well-scoped: the scored coverage is identical either way, so it only selects the bonus capability catalog — documented in `--help` and the README. Criterion 5's "`--store`, `--help`" wording was updated to record it. + +- **`DEFAULT_STORE_ROOT` moved to `store.ts`.** It was defined in `run.ts`; its natural home is the store, and both commands now need it, so it lives in `store.ts` as the single source of truth and is re-exported from `run.ts` to keep that command's existing importers (and `run.test.ts`) resolving it unchanged. + +- **Tests, TDD.** `bench/report.test.ts` (8 tests) was authored test-first by the test-writer sub-agent from the public interface alone: the parser defaults/overrides/rejection/help, and the command over populated/empty/help paths. As with the `run-loop` slice, the parser is a small cohesive unit whose logic all landed in one GREEN, so the parser cases past the first are passing characterization guards rather than red-first cycles; the `runReportCommand` cell had a genuine red first. Full bench tier green (104 tests), typecheck clean. + +- **Review (three-axis, `/review-uncommitted`): Risk overall Low.** Spec axis: all five criteria met; the only finding was that `--self-review` exceeded the written seam — justified in substance, now recorded here and in the criterion. Standards axis: no hard violations; two judgement calls — the `parseReportArgs` flag loop partly overlaps `parseRunArgs` but has genuinely diverged (adds `--no-` negation), left unextracted as the reviewer advised (a shared parser would be premature); and the unused `deps` parameter, addressed with a parity comment. diff --git a/bench/README.md b/bench/README.md index 2d44563..1e7fac9 100644 --- a/bench/README.md +++ b/bench/README.md @@ -50,8 +50,7 @@ The raw component breakdown is retained on every sample so the data can be re-we - `run-loop.ts` — `runCells`, the run loop: it runs one chosen `(arm, task)` cell for a batch of trials (defaulting to five, with a reporting floor of three) by driving `runCell` and the sample store, deciding only how many trials to run and at what trial numbers. Deepening a cell continues numbering past the highest trial it already holds and appends, so a cell's sample size grows across sittings rather than being overwritten. It reimplements no orchestration — provision, run, score, and append stay in `runCell`. - `run.ts` — the maintainer-facing run-loop command. `parseRunArgs` is the pure, unit-tested argument seam; `runBenchCommand` is the live boundary that resolves host access, resolves the scored suite against the host's self-review support, selects the task, and drives `runCells`. Invoked via `npm run bench:run` (see below). - `aggregate.ts` — the aggregator: the pure seam that renders the accumulated sample store into a readable comparison. `readAllSamples` drains a store into a flat record list; `aggregate` rolls the records up against the task definitions into a `Report` (headline, per-tier and per-token-component breakdowns, and the bonus table); `renderReport` renders that report as stable text. The headline metric, `costEquivalentTokens`, is computed here at render time by weighting the four retained token components by ADR 0014's pricing ratios (fresh input 1×, cache-write 1.25×, cache-read 0.1×, output 5×), so the stored records can be re-weighted without re-running. It reads whatever samples exist and annotates incomplete coverage — cells below the reporting floor (partial) and unsampled cells (missing) — rather than blocking on a complete matrix. A pure function of the records plus the definitions, unit-tested against synthetic append-only sample stores. - -The aggregator has no command wrapper yet; a `bench:report` CLI over it is a natural follow-up, in the same shape as `run.ts` is over `run-loop.ts`. +- `report.ts` — the maintainer-facing reporting command, the counterpart to `run.ts`. `parseReportArgs` is the pure argument seam; `runReportCommand` opens the store, drains it, aggregates against the scored suite and bonus, and prints `renderReport`. Unlike `run.ts` it has no live boundary — it reads only the local store on disk (no credentials, host, or Agent SDK) — so the whole command is deterministic and unit-tested, not smoke-run. Invoked via `npm run bench:report` (see below). ## Running a cell @@ -67,6 +66,20 @@ Re-running the same cell deepens it — the new trials append rather than overwr The command is executed with [`tsx`](https://tsx.is) (a devDependency) so the harness's TypeScript runs directly. Like the runner smoke tier it drives the live host and the Claude Agent SDK, so a real run needs a configured login, the `gitea-axi` CLI on `PATH`, the Agent SDK installed (`npm install @anthropic-ai/claude-agent-sdk` — an optional peer, declared but not installed by default), and a Claude subscription. +## Reading the results + +The reporting command renders whatever has accumulated in the store into the readable comparison — the cost-equivalent-token headline, coverage annotated against the reporting floor, the per-tier and per-token-component breakdowns, and the bonus table: + +``` +npm run bench:report +``` + +It reads `bench/results/` by default; pass `--store ` to read a different store, and `--help` for the full flag list. +Incomplete coverage is annotated rather than hidden, so a half-run matrix still renders (unrun arms show an em dash, not a misleading zero). + +Unlike `bench:run` this command is offline — it reads only the local store, never the host — so it needs no login, host, or Agent SDK, and is safe to run at any time. +The `--self-review` / `--no-self-review` flag only selects the bonus capability catalog (whether the approve/request-changes review pair appears there or in the scored suite); the scored coverage is identical either way, so set it to match the host the samples were run on. + ## Tests The harness's deterministic seams are unit-tested in this directory, colocated with their source, and run via: diff --git a/bench/report.test.ts b/bench/report.test.ts new file mode 100644 index 0000000..1265762 --- /dev/null +++ b/bench/report.test.ts @@ -0,0 +1,177 @@ +import { mkdtempSync, rmSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { afterEach, beforeEach, describe, expect, it } from "vitest"; +import type { CliDeps } from "../src/deps.js"; +import type { ResultRecord } from "./result.js"; +import { createSampleStore, DEFAULT_STORE_ROOT } from "./store.js"; +import { parseReportArgs, runReportCommand } from "./report.js"; + +/** A minimal passing gitea-axi ResultRecord, overridable per sample. */ +function record(overrides: Partial = {}): ResultRecord { + return { + arm: "gitea-axi", + taskId: "t1", + tier: "read", + trial: 1, + timestamp: "2026-07-16T00:00:00Z", + tokens: { freshInput: 0, cacheCreation: 0, cacheRead: 0, output: 0 }, + turns: 0, + durationMs: 0, + imputedCostUsd: 0, + outcome: { pass: true }, + ...overrides, + }; +} + +describe("parseReportArgs", () => { + // Behavior: with no arguments the parser resolves the documented defaults — + // the store root falls back to the module's DEFAULT_STORE_ROOT constant (the + // single source of truth for that default, so we assert against the imported + // constant rather than the hardcoded "bench/results" string, checking the + // parser wires the module default through on omission), and self-review + // defaults to permitted, the richer scored variant (independent literal true). + it("resolves the documented defaults when no arguments are given", () => { + const result = parseReportArgs([]); + + expect(result.help).toBe(false); + if (result.help) return; + + expect(result.storeRoot).toBe(DEFAULT_STORE_ROOT); + expect(result.selfReview).toBe(true); + }); + + // Behavior: the spaced value form of --store overrides the store root the report + // reads, and --no-self-review flips the self-review variant off. Expected values + // are independent literals: the store root the maintainer supplied and false. + it("overrides the store root via spaced --store and disables self-review via --no-self-review", () => { + const result = parseReportArgs(["--store", "/tmp/bench-out", "--no-self-review"]); + + expect(result.help).toBe(false); + if (result.help) return; + + expect(result.storeRoot).toBe("/tmp/bench-out"); + expect(result.selfReview).toBe(false); + }); + + // Behavior: --store also accepts the inline --store= form, and --self-review + // states the default explicitly (permitted). Expected values are independent + // literals: the inline store root and true. + it("accepts the inline --store= form and honors an explicit --self-review", () => { + const result = parseReportArgs(["--store=/tmp/x", "--self-review"]); + + expect(result.help).toBe(false); + if (result.help) return; + + expect(result.storeRoot).toBe("/tmp/x"); + expect(result.selfReview).toBe(true); + }); + + // Behavior: malformed input is rejected with a usage error — an unknown flag, a + // bare positional argument (not a --flag), a value flag (--store) missing its + // value, and a value handed to the boolean --self-review (which takes none). + it("rejects malformed input with a usage error", () => { + // Unknown flag. + expect(() => parseReportArgs(["--frobnicate", "x"])).toThrow(); + // Bare positional argument, not a --flag. + expect(() => parseReportArgs(["positional"])).toThrow(); + // Value flag missing its value. + expect(() => parseReportArgs(["--store"])).toThrow(); + // The boolean --self-review takes no value. + expect(() => parseReportArgs(["--self-review=yes"])).toThrow(); + }); + + // Behavior: --help and its -h alias short-circuit parsing to a help request, + // winning even alongside other arguments. + it("short-circuits to a help request for --help and -h, even alongside other args", () => { + expect(parseReportArgs(["--help"]).help).toBe(true); + expect(parseReportArgs(["-h"]).help).toBe(true); + expect(parseReportArgs(["--store", "/tmp/x", "--help"]).help).toBe(true); + }); +}); + +describe("runReportCommand", () => { + let root: string; + + beforeEach(() => { + root = mkdtempSync(join(tmpdir(), "bench-report-")); + }); + + afterEach(() => { + rmSync(root, { recursive: true, force: true }); + }); + + // Behavior: the offline reporting boundary opens the sample store at --store, + // drains it, aggregates against the scored suite and bonus, and prints the + // rendered comparison line-by-line through `out`, returning exit code 0. Given + // three passing gitea-axi samples on one read task — reaching the reporting + // floor of three — the rendered output labels the headline metric + // ("cost-equivalent", per the spec / ADR 0014), gives every arm a row (the + // four arms are gitea-axi, tea, gitea-mcp, raw-api), and annotates coverage + // against the "reporting floor of 3". Assertions are on the exit code and the + // presence of content, never on column layout, so they survive a rendering + // refactor. The command reads no credentials, host, or SDK, so deps are empty. + it("opens the store, aggregates, and prints the rendered comparison with exit code 0", async () => { + const store = createSampleStore(root); + store.append(record({ arm: "gitea-axi", taskId: "t1", tier: "read", trial: 1 })); + store.append(record({ arm: "gitea-axi", taskId: "t1", tier: "read", trial: 2 })); + store.append(record({ arm: "gitea-axi", taskId: "t1", tier: "read", trial: 3 })); + + const deps: CliDeps = { env: {}, cwd: "/", globals: {} }; + const lines: string[] = []; + const code = await runReportCommand(["--store", root], deps, (l) => lines.push(l)); + const output = lines.join("\n"); + + // Success exit code. + expect(code).toBe(0); + + // The headline metric is labelled. + expect(output.toLowerCase()).toContain("cost-equivalent"); + + // Every arm gets a row. + expect(output).toContain("gitea-axi"); + expect(output).toContain("tea"); + + // Coverage is annotated against the reporting floor of three, which the + // three samples reach. + expect(output).toContain("reporting floor of 3"); + }); + + // Behavior: an empty (never-written) store renders without error, inheriting + // the aggregator's placeholder behavior rather than special-casing it. The + // headline still labels the metric and every arm, and marks the unrun arms + // with the em-dash placeholder ("—", U+2014) rather than a misleading zero — + // the established renderReport behavior pinned by bench/aggregate.test.ts. + // Returns exit code 0. The `root` from beforeEach is created empty; nothing is + // appended. + it("renders an empty store without error, marking unrun arms with the em-dash placeholder", async () => { + const deps: CliDeps = { env: {}, cwd: "/", globals: {} }; + const lines: string[] = []; + const code = await runReportCommand(["--store", root], deps, (l) => lines.push(l)); + const output = lines.join("\n"); + + // Success exit code. + expect(code).toBe(0); + + // The headline metric is labelled and every arm still appears. + expect(output.toLowerCase()).toContain("cost-equivalent"); + expect(output).toContain("gitea-axi"); + + // An unrun arm shows the em-dash placeholder, never a zero. + expect(output).toContain("—"); + }); + + // Behavior: --help prints the command's help and returns 0 without reading any + // store, so it works before a store exists. No store path is created or + // referenced. The help text names the command ("bench:report") — an + // independent literal. + it("prints help naming the command and returns 0 without reading a store", async () => { + const deps: CliDeps = { env: {}, cwd: "/", globals: {} }; + const lines: string[] = []; + const code = await runReportCommand(["--help"], deps, (l) => lines.push(l)); + const output = lines.join("\n"); + + expect(code).toBe(0); + expect(output).toContain("bench:report"); + }); +}); diff --git a/bench/report.ts b/bench/report.ts new file mode 100644 index 0000000..800955b --- /dev/null +++ b/bench/report.ts @@ -0,0 +1,171 @@ +// The maintainer-facing reporting command: render the accumulated sample store +// into the readable comparison. +// +// This is the reporting counterpart to run.ts. Where the run command spends the +// token budget on one cell, this command reads whatever samples have accumulated +// so far and prints the aggregator's comparison — headline, coverage, per-tier +// and per-token-component breakdowns, and the bonus table. +// +// Unlike run.ts, this command has no live boundary: it touches only the local +// sample store on disk (no credentials, host, or Agent SDK), so the whole command +// is deterministic and unit-tested. `parseReportArgs` is the pure argument seam. + +import { pathToFileURL } from "node:url"; +import type { CliDeps } from "../src/deps.js"; +import { aggregate, readAllSamples, renderReport } from "./aggregate.js"; +import { createSampleStore, DEFAULT_STORE_ROOT } from "./store.js"; +import { buildBonusTasks, buildScoredSuite } from "./task-suite.js"; + +/** A fully-resolved report configuration. */ +export interface ReportArgs { + /** The sample-store root to read; defaults to {@link DEFAULT_STORE_ROOT}. */ + storeRoot: string; + /** Whether to render the suite/bonus variant for a self-review-permitting host. */ + selfReview: boolean; +} + +/** The parse outcome: a request for help, or a resolved configuration to render. */ +export type ParsedReportArgs = { help: true } | ({ help: false } & ReportArgs); + +/** The value-taking flags the command understands. */ +const VALUE_FLAGS = new Set(["store"]); + +/** The boolean flags the command understands, each with a `--no-` negation. */ +const BOOLEAN_FLAGS = new Set(["self-review"]); + +/** A usage error, surfaced to the maintainer with the offending detail. */ +function usage(detail: string): Error { + return new Error(`${detail}\n\nUsage: bench:report [--store ] [--self-review | --no-self-review]`); +} + +/** + * Parse the report command's argv into a resolved configuration, applying + * defaults ({@link DEFAULT_STORE_ROOT} store, self-review permitted). A report + * needs no required selection, so no arguments is a valid invocation. Throws a + * usage error on an unknown flag, a bare argument, a value-flag missing its + * value, or a value handed to a boolean flag. + */ +export function parseReportArgs(argv: string[]): ParsedReportArgs { + let storeRoot = DEFAULT_STORE_ROOT; + let selfReview = true; + + for (let index = 0; index < argv.length; index += 1) { + const token = argv[index] as string; + if (token === "--help" || token === "-h") { + return { help: true }; + } + if (!token.startsWith("--")) { + throw usage(`unexpected argument "${token}"`); + } + + const equals = token.indexOf("="); + const rawName = equals === -1 ? token.slice(2) : token.slice(2, equals); + const inlineValue = equals === -1 ? undefined : token.slice(equals + 1); + + // A `--no-` prefix negates a boolean flag. + const negated = rawName.startsWith("no-"); + const name = negated ? rawName.slice(3) : rawName; + + if (BOOLEAN_FLAGS.has(name)) { + if (inlineValue !== undefined) { + throw usage(`flag --${rawName} takes no value`); + } + selfReview = !negated; + continue; + } + + if (negated || !VALUE_FLAGS.has(name)) { + throw usage(`unknown flag "--${rawName}"`); + } + + let value = inlineValue; + if (value === undefined) { + value = argv[index + 1]; + if (value === undefined || value.startsWith("--")) { + throw usage(`flag --${name} needs a value`); + } + index += 1; + } + if (name === "store") { + storeRoot = value; + } + } + + return { help: false, storeRoot, selfReview }; +} + +/** The help text printed for `--help` / `-h`. */ +const HELP_TEXT = `bench:report — render the accumulated benchmark samples into a comparison. + +Reads whatever samples have accumulated in the store and prints the aggregator's +comparison: the cost-equivalent-token headline, coverage annotated against the +reporting floor, per-tier and per-token-component breakdowns, and the bonus table. +Incomplete coverage is annotated rather than hidden, so a half-run matrix still +renders. This command is offline — it reads only the local store, never the host. + +Usage: + npm run bench:report -- [options] + +Options: + --store Sample store root to read (default: ${DEFAULT_STORE_ROOT}) + --self-review Render the variant for a self-review-permitting host (default) + --no-self-review Render the variant for a host that forbids self-review + -h, --help Show this help + +--self-review only affects the bonus capability catalog (whether the approve / +request-changes review pair appears there or in the scored suite); the scored +coverage is identical either way. Set it to match the host the samples were run on.`; + +/** + * Render the accumulated sample store into the readable comparison. This is the + * command's boundary, but — unlike the run command — it is offline: it opens the + * local store at `--store`, drains it, aggregates against the scored suite and + * bonus definitions (resolved against the `--self-review` variant), and prints the + * rendered report through `out`. No orchestration, weighting, or rendering is + * reimplemented here; it drives the `readAllSamples` / `aggregate` / `renderReport` + * seam. Returns a process exit code. + */ +export async function runReportCommand( + argv: string[], + // Unused: an offline report needs no credentials, cwd, or env. Kept for signature + // parity with the command family (runBenchCommand takes the same (argv, deps, out)). + _deps: CliDeps, + out: (line: string) => void, +): Promise { + const parsed = parseReportArgs(argv); + if (parsed.help) { + out(HELP_TEXT); + return 0; + } + + const suiteOptions = { selfReviewPermitted: parsed.selfReview }; + const store = createSampleStore(parsed.storeRoot); + const report = aggregate({ + records: readAllSamples(store), + suite: buildScoredSuite(suiteOptions), + bonus: buildBonusTasks(suiteOptions), + }); + out(renderReport(report)); + return 0; +} + +/** Entry point: render the report and set the process exit code. */ +export async function main(): Promise { + const deps: CliDeps = { env: process.env, cwd: process.cwd(), globals: {} }; + try { + process.exitCode = await runReportCommand( + process.argv.slice(2), + deps, + (line) => process.stdout.write(`${line}\n`), + ); + } catch (error) { + process.stderr.write(`${error instanceof Error ? error.message : String(error)}\n`); + process.exitCode = 1; + } +} + +// Run only when executed directly (e.g. `tsx bench/report.ts`), not when imported +// by a test. Under a TypeScript runner argv[1] is this file's own path. +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + await main(); +} diff --git a/bench/run.ts b/bench/run.ts index b4980e5..e92913c 100644 --- a/bench/run.ts +++ b/bench/run.ts @@ -27,18 +27,19 @@ import { DEFAULT_TRIALS, REPORTING_FLOOR, runCells, type RunCellsResult } from " import { resolveBenchAccess } from "./seed.js"; import { detectSelfReviewSupport } from "./self-review.js"; import { sdkAgentDriver } from "./sdk-driver.js"; -import { createSampleStore } from "./store.js"; +import { createSampleStore, DEFAULT_STORE_ROOT } from "./store.js"; import { buildScoredSuite } from "./task-suite.js"; +// Re-exported so this command's existing importers keep resolving it from here; +// its single source of truth is now the store, which owns the default root. +export { DEFAULT_STORE_ROOT }; + /** The default turn cap a run is bounded by when not overridden. */ export const DEFAULT_TURN_CAP = 40; /** The default wall-clock backstop (ms) a run is bounded by when not overridden. */ export const DEFAULT_WALL_CLOCK_MS = 300_000; -/** Where accumulated samples are stored when `--store` is not given. */ -export const DEFAULT_STORE_ROOT = "bench/results"; - /** Environment variable naming the tea login the benchmark authenticates through. */ export const LOGIN_ENV = "GITEA_AXI_BENCH_LOGIN"; diff --git a/bench/store.ts b/bench/store.ts index cfea61a..5e1e084 100644 --- a/bench/store.ts +++ b/bench/store.ts @@ -2,6 +2,9 @@ import { appendFileSync, mkdirSync, readdirSync, readFileSync } from "node:fs"; import { dirname, join } from "node:path"; import type { CellKey, ResultRecord } from "./result.js"; +/** Where accumulated samples are stored when no store root is given. */ +export const DEFAULT_STORE_ROOT = "bench/results"; + /** Run `read`, returning `fallback` when the target does not exist yet. */ function ignoreEnoent(read: () => T, fallback: T): T { try { diff --git a/package.json b/package.json index c6bb2f2..f13f638 100644 --- a/package.json +++ b/package.json @@ -37,7 +37,8 @@ "test:pack": "vitest run --config vitest.packaging.config.ts", "test:bench": "vitest run --config vitest.bench.config.ts", "test:bench:smoke": "vitest run --config vitest.bench-smoke.config.ts", - "bench:run": "tsx bench/run.ts" + "bench:run": "tsx bench/run.ts", + "bench:report": "tsx bench/report.ts" }, "dependencies": { "@toon-format/toon": "^2.3.0",