diff --git a/bench/checker.test.ts b/bench/checker.test.ts index ddaebe7..4f3fa36 100644 --- a/bench/checker.test.ts +++ b/bench/checker.test.ts @@ -340,6 +340,64 @@ describe("checkReadAnswer", () => { expect(result.pass).toBe(false); }); + + it("passes via pattern when filler words break every contiguous anyOf phrase", () => { + // A count fact is brittle when pinned to fixed phrases: a human padding the + // answer with filler words ("issues are currently") splits any contiguous + // rendering. The optional `pattern` — a case-insensitive regex source — spans + // that filler as a bounded run of alphabetic words between the number and + // "open", so the fact is satisfied even though no `anyOf` phrase appears verbatim. + const facts: RequiredFact[] = [ + { + description: "count of open issues", + anyOf: ["5 open", "5 issues are open"], + pattern: "\\b5(?: [a-z]+){0,4} open\\b", + }, + ]; + + // "issues are currently" is inserted between "5" and "open", so neither + // contiguous anyOf phrase matches — but the pattern's alphabetic filler run does. + const report = "5 issues are currently open (5 of 5 total)."; + + expect(checkReadAnswer(facts, report)).toEqual({ pass: true }); + }); + + it("fails when a digit interrupts the number-to-open run, so a wrong count cannot slip through", () => { + // The same pattern-bearing fact as above. Here the report states a DIFFERENT + // open count (3), merely mentioning the number 5 elsewhere. The pattern's + // filler run is alphabetic only, so the digit "3" between the matched "5" and + // "open" is not spanned, and no anyOf phrase matches either — the fact fails. + const facts: RequiredFact[] = [ + { + description: "count of open issues", + anyOf: ["5 open", "5 issues are open"], + pattern: "\\b5(?: [a-z]+){0,4} open\\b", + }, + ]; + + // "5 issues in total, and 3 open" — the 5 is a total, the open count is 3. + const report = "There are 5 issues in total, and 3 open."; + + const result = checkReadAnswer(facts, report); + + expect(result.pass).toBe(false); + // The unmet fact must remain identifiable by its own description. + if (result.pass === false) { + expect(result.differences.some((d) => d.includes("count of open issues"))).toBe(true); + } + }); + + it("still matches on anyOf alone when a fact carries no pattern", () => { + // A fact without `pattern` behaves exactly as before: only the contiguous + // anyOf renderings are consulted, unaffected by the new pattern support. + const facts: RequiredFact[] = [ + { description: "count of open issues", anyOf: ["7 open issues", "seven open issues"] }, + ]; + + const report = "The board shows 7 open issues right now."; + + expect(checkReadAnswer(facts, report)).toEqual({ pass: true }); + }); }); describe("score", () => { diff --git a/bench/checker.ts b/bench/checker.ts index c51a217..198a0ed 100644 --- a/bench/checker.ts +++ b/bench/checker.ts @@ -241,9 +241,7 @@ function matchByKey( */ export function checkReadAnswer(facts: RequiredFact[], report: string): CheckResult { const haystack = normalizeText(report); - const missing = facts.filter( - (fact) => !fact.anyOf.some((rendering) => haystack.includes(normalizeText(rendering))), - ); + const missing = facts.filter((fact) => !factPresent(fact, haystack)); if (missing.length === 0) { return { pass: true }; } @@ -253,6 +251,20 @@ export function checkReadAnswer(facts: RequiredFact[], report: string): CheckRes }; } +/** + * Whether a required fact is present in the normalized report: any `anyOf` + * rendering as a contiguous substring, or the optional `pattern` regex matching. + * The pattern is compiled case-insensitively over the already-normalized text, so + * it recognises a count padded with filler ("5 issues are currently open") that no + * fixed phrase would. + */ +function factPresent(fact: RequiredFact, haystack: string): boolean { + if (fact.anyOf.some((rendering) => haystack.includes(normalizeText(rendering)))) { + return true; + } + return fact.pattern !== undefined && new RegExp(fact.pattern, "i").test(haystack); +} + /** * Lower-case, drop markdown emphasis/code markers, and collapse runs of * whitespace so incidental phrasing and formatting do not matter — a report that diff --git a/bench/scoring-spec.ts b/bench/scoring-spec.ts index c8d9638..82dda8e 100644 --- a/bench/scoring-spec.ts +++ b/bench/scoring-spec.ts @@ -131,10 +131,21 @@ export interface RepoState { * whitespace and case normalization), so a count or a name can be phrased * variously without resorting to an LLM judge. `description` names the fact in * diagnostics when it is missing. + * + * `anyOf` matches a *contiguous* substring, which is brittle for facts a human + * naturally pads with filler — "5 issues are currently open" does not contain + * the fixed phrase "5 issues are open". For those, supply `pattern`: a regular + * expression (matched against the same normalized report) that satisfies the + * fact when it matches, so the count itself can be recognised rather than one + * exact wording. A fact is present when any `anyOf` rendering *or* `pattern` + * matches; `anyOf` stays the human-readable renderings even when `pattern` + * carries the real matcher. */ export interface RequiredFact { description: string; anyOf: string[]; + /** Optional regex (source, matched case-insensitively against the normalized report). */ + pattern?: string; } /** diff --git a/bench/seed.ts b/bench/seed.ts index 7ad9a3b..531ef00 100644 --- a/bench/seed.ts +++ b/bench/seed.ts @@ -359,12 +359,122 @@ async function ensurePullRequest( await ensureReviews(access, coords, number, pr.reviews); } +/** How long the readiness gate polls before giving up, and how often it re-checks. */ +const READINESS_TIMEOUT_MS = 60_000; +const READINESS_INTERVAL_MS = 750; + +/** Resolve after `ms` milliseconds. */ +function delay(ms: number): Promise { + return new Promise((resolve) => setTimeout(resolve, ms)); +} + +/** A search hit's minimal shape: the repository the matched issue/pull belongs to. */ +interface SearchHit { + repository?: { name?: string } | null; +} + +/** + * Whether the issue indexer has caught up: a full-text search for `title` of the + * given kind returns a hit in this repository. The endpoint spans every repo the + * owner can access (mirroring `gitea-axi search`), so a hit only counts when its + * repository matches. A non-2xx or a not-yet-indexed title reads as not-ready. + */ +async function searchIndexed( + access: BenchAccess, + coords: RepoCoords, + type: "issues" | "pulls", + title: string, +): Promise { + const query = new URLSearchParams({ q: title, type, owner: coords.owner, state: "all", limit: "50" }); + const res = await request(access, "GET", `/repos/issues/search?${query.toString()}`); + if (!res.ok) { + return false; + } + const hits = (await res.json()) as SearchHit[]; + return hits.some((hit) => hit.repository?.name === coords.repo); +} + +/** + * The reason a freshly seeded repository is not yet ready for the agent, or `null` + * when it is. Ready means an independent read sees the repository and its full + * seeded issue and pull spread, and the issue indexer returns a seeded issue and + * pull — the two consistency windows (repo visibility and index lag) an agent + * would otherwise race and fail against. Every read is non-throwing, so a + * transient error reads as not-ready and is retried rather than propagated. + */ +async function readinessGap(access: BenchAccess, coords: RepoCoords): Promise { + const repoPath = `/repos/${coords.owner}/${coords.repo}`; + const repoRes = await request(access, "GET", repoPath); + if (!repoRes.ok) { + return `repository read returned ${repoRes.status}`; + } + const issuesRes = await request(access, "GET", `${repoPath}/issues?type=issues&state=all&limit=100`); + if (!issuesRes.ok) { + return `issue list returned ${issuesRes.status}`; + } + const issues = (await issuesRes.json()) as GiteaIssue[]; + if (issues.length < SEED_PLAN.issues.length) { + return `only ${issues.length}/${SEED_PLAN.issues.length} issues visible`; + } + const pullsRes = await request(access, "GET", `${repoPath}/pulls?state=all&limit=100`); + if (!pullsRes.ok) { + return `pull list returned ${pullsRes.status}`; + } + const pulls = (await pullsRes.json()) as GiteaPull[]; + if (pulls.length < SEED_PLAN.pullRequests.length) { + return `only ${pulls.length}/${SEED_PLAN.pullRequests.length} pull requests visible`; + } + const sampleIssue = SEED_PLAN.issues[0]!.title; + if (!(await searchIndexed(access, coords, "issues", sampleIssue))) { + return `issue "${sampleIssue}" not yet indexed for search`; + } + const samplePull = SEED_PLAN.pullRequests[0]!.title; + if (!(await searchIndexed(access, coords, "pulls", samplePull))) { + return `pull request "${samplePull}" not yet indexed for search`; + } + return null; +} + +/** + * Block until a freshly seeded repository is fully consistent for the agent, or + * fail loudly if it never settles within the timeout. Gitea makes a just-created + * repository and its just-written issues and pulls visible to the seed's own + * writes immediately, but an independent reader — the agent, a fresh process + * moments later — can hit a brief 404 window on the repository and a longer lag on + * the async issue indexer. Both are the benchmark's races, not the tool's, so + * closing them here keeps a scored run measuring gitea-axi rather than host + * propagation. A timeout is a harness failure surfaced to the maintainer, never a + * scored agent failure. + */ +export async function waitForSeedReady( + access: BenchAccess, + coords: RepoCoords, + timeoutMs = READINESS_TIMEOUT_MS, + intervalMs = READINESS_INTERVAL_MS, +): Promise { + const deadline = Date.now() + timeoutMs; + let gap = await readinessGap(access, coords); + while (gap !== null) { + if (Date.now() >= deadline) { + throw new Error( + `seeded repository ${coords.owner}/${coords.repo} not ready after ${timeoutMs}ms: ${gap}`, + ); + } + await delay(intervalMs); + gap = await readinessGap(access, coords); + } +} + /** * Seed a freshly provisioned repository to the ground truth, idempotently. Labels * come first (so issues and pull requests can apply them), then the issue spread, * then the pull requests. Returns the deterministic ground-truth RepoState the * checker scores against; on a fresh repository the created numbers match it, * and a re-run leaves them unchanged. + * + * Before returning, it waits out the repository-visibility and issue-indexer + * consistency windows (see waitForSeedReady), so the agent that runs next never + * races a repository that is not yet readable or searchable. */ export async function seedRepo(access: BenchAccess, coords: RepoCoords): Promise { const user = await currentUser(access); @@ -384,6 +494,7 @@ export async function seedRepo(access: BenchAccess, coords: RepoCoords): Promise await ensurePullRequest(access, coords, pr, labelIds, pullsByTitle); } + await waitForSeedReady(access, coords); return groundTruth(user); } diff --git a/bench/task-suite.ts b/bench/task-suite.ts index 832e1f4..bbd25ad 100644 --- a/bench/task-suite.ts +++ b/bench/task-suite.ts @@ -92,6 +92,11 @@ function readTasks(): BenchTask[] { { description: "the repository has 5 open issues", anyOf: ["5 open", "five open", "open issues: 5", "open: 5", "5 issues are open"], + // "5" followed by "open" across up to four alphabetic filler words, so + // natural padding ("5 issues are currently open") is recognised while a + // digit between them (a wrong "5 issues, 3 open") is not: the filler run + // is alphabetic only, so it cannot span another count. + pattern: "\\b5(?: [a-z]+){0,4} open\\b", }, ]), },