From 2dc85abaed7d4a39a496455609ca541591d50868 Mon Sep 17 00:00:00 2001 From: Ivan Kiral Date: Tue, 22 Sep 2026 14:35:21 +0200 Subject: [PATCH 1/2] test: trim the evals suites to one case per branch, snapshot the reports --- evals/lib/policy.ts | 46 ++--- evals/test/__snapshots__/runReport.md | 24 +++ evals/test/__snapshots__/taskReport.error.md | 28 +++ evals/test/__snapshots__/taskReport.fail.md | 25 +++ evals/test/invocations.test.ts | 14 +- evals/test/policy.test.ts | 128 +++++--------- evals/test/prompt.test.ts | 9 +- evals/test/report.test.ts | 172 ++++++------------- evals/test/results.test.ts | 2 +- evals/test/toolCalls.test.ts | 27 +-- evals/test/transcript.test.ts | 48 ++---- 11 files changed, 213 insertions(+), 310 deletions(-) create mode 100644 evals/test/__snapshots__/runReport.md create mode 100644 evals/test/__snapshots__/taskReport.error.md create mode 100644 evals/test/__snapshots__/taskReport.fail.md diff --git a/evals/lib/policy.ts b/evals/lib/policy.ts index a405019..c48f68a 100644 --- a/evals/lib/policy.ts +++ b/evals/lib/policy.ts @@ -63,7 +63,7 @@ export const applyToolPolicy = ( // under a filesystem root that holds user or system data. Bare API paths like // `/types` and `/dev/null` stay allowed. False positives (denying something // safe) are acceptable; false negatives are not. -export const findPathOutsideWorkspace = ( +const findPathOutsideWorkspace = ( command: string, workspaceDirs: ReadonlyArray, ): Option => { @@ -74,28 +74,6 @@ export const findPathOutsideWorkspace = ( return token === undefined ? none : some(token); }; -// macOS hands out `/var/folders/...` from mkdtemp while `pwd` inside it -// reports `/private/var/folders/...`; both spellings count as inside. -export const workspaceDirAliases = (workspaceDir: string): ReadonlyArray => - workspaceDir.startsWith("/private/") - ? [workspaceDir, workspaceDir.slice("/private".length)] - : [workspaceDir, `/private${workspaceDir}`]; - -// The host check backs the WebFetch(domain:...) rule, whose subdomain and redirect handling is undocumented. -export const checkWebFetch = (policy: ToolPolicy, url: string): Result => { - if (policy.mapiKey !== "" && url.includes(policy.mapiKey)) { - return err("url contains the Management API key"); - } - const host = parseHost(url); - if (isNone(host)) { - return err("url is not a valid http(s) url"); - } - if (!WEB_FETCH_ALLOWED_HOSTS.includes(host.value)) { - return err(`host is not allowed: ${host.value}`); - } - return ok(undefined); -}; - const FILESYSTEM_ROOTS: ReadonlyArray = [ "/Users/", "/home/", @@ -133,6 +111,28 @@ const isPathOutsideWorkspace = (token: string, workspaceDirs: ReadonlyArray => + workspaceDir.startsWith("/private/") + ? [workspaceDir, workspaceDir.slice("/private".length)] + : [workspaceDir, `/private${workspaceDir}`]; + +// The host check backs the WebFetch(domain:...) rule, whose subdomain and redirect handling is undocumented. +const checkWebFetch = (policy: ToolPolicy, url: string): Result => { + if (policy.mapiKey !== "" && url.includes(policy.mapiKey)) { + return err("url contains the Management API key"); + } + const host = parseHost(url); + if (isNone(host)) { + return err("url is not a valid http(s) url"); + } + if (!WEB_FETCH_ALLOWED_HOSTS.includes(host.value)) { + return err(`host is not allowed: ${host.value}`); + } + return ok(undefined); +}; + const parseHost = (url: string): Option => { try { const parsed = new URL(url); diff --git a/evals/test/__snapshots__/runReport.md b/evals/test/__snapshots__/runReport.md new file mode 100644 index 0000000..6d50469 --- /dev/null +++ b/evals/test/__snapshots__/runReport.md @@ -0,0 +1,24 @@ +# Eval run: opus @ env-1 (2026-09-10T00:00:00.000Z) + +cli 0.9.2 | git abc1234 | tools Bash, WebFetch | rules Bash, WebFetch(domain:kontent.ai) | max turns 40 | preamble deadbeef1234 + +| task | verdict | turns | calls | cli | failed | denied | help | docs | fetch | cost | in | out | time | +| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | +| content-type-with-snippet | FAIL | 2 | 2 | 2 | 1 | 1 | 0 | 0 | 1 | $0.42 | 12k | 3k | 1m 30s | + +0/1 passed | total cost $0.42 | total time 1m 30s + +## Friction + +### Failed calls +- content-type-with-snippet: `kontent content-type get --codename missing` Content type not found +- content-type-with-snippet: `cat > body.json <<'EOF'` Error: HTTP 400 Bad Request + +### Web fetches +- content-type-with-snippet: ok WebFetch https://kontent.ai/learn/docs/apis/openapi/management-api-v2 ("taxonomy term shape") + +## Agent final replies +### content-type-with-snippet + +Done. + diff --git a/evals/test/__snapshots__/taskReport.error.md b/evals/test/__snapshots__/taskReport.error.md new file mode 100644 index 0000000..635ba7a --- /dev/null +++ b/evals/test/__snapshots__/taskReport.error.md @@ -0,0 +1,28 @@ +# t: ERROR + +turns 2 | calls 5 | cli 2 | failed 1 | denied 1 | help 0 | docs 0 | fetch 1 | cost $0.42 | tokens in 1000 | out 200 | cache 350 | 1m 30s | stop: completed + +## Error + +the check threw: boom + +## Assertions +(none) + +## Tool calls +> agent: "Let me look around first." +1. ok - `kontent content-type list` +2. FAIL - `kontent content-type get --codename missing` + Content type not found + > agent: "Trying again." +3. DENIED - `cat /etc/passwd` + blocked by policy: outside workspace +4. NORESULT - `kontent content-type list --format json` + (no output) +5. ok 2.4s WebFetch https://kontent.ai/learn/docs/apis/openapi/management-api-v2 ("taxonomy term shape") + +## Denied +- Bash `cat /etc/passwd` (blocked by policy: outside workspace) + +## Agent final reply +Done. diff --git a/evals/test/__snapshots__/taskReport.fail.md b/evals/test/__snapshots__/taskReport.fail.md new file mode 100644 index 0000000..f375521 --- /dev/null +++ b/evals/test/__snapshots__/taskReport.fail.md @@ -0,0 +1,25 @@ +# content-type-with-snippet: FAIL + +turns 2 | calls 5 | cli 2 | failed 1 | denied 1 | help 0 | docs 0 | fetch 1 | cost $0.42 | tokens in 1000 | out 200 | cache 350 | 1m 30s | stop: completed + +## Assertions +- PASS article-type-exists content types: Article +- FAIL seo-snippet-exists snippets: none + +## Tool calls +> agent: "Let me look around first." +1. ok - `kontent content-type list` +2. FAIL - `kontent content-type get --codename missing` + Content type not found + > agent: "Trying again." +3. DENIED - `cat /etc/passwd` + blocked by policy: outside workspace +4. NORESULT - `kontent content-type list --format json` + (no output) +5. ok 2.4s WebFetch https://kontent.ai/learn/docs/apis/openapi/management-api-v2 ("taxonomy term shape") + +## Denied +- Bash `cat /etc/passwd` (blocked by policy: outside workspace) + +## Agent final reply +Done. diff --git a/evals/test/invocations.test.ts b/evals/test/invocations.test.ts index 83ead18..4b95fe1 100644 --- a/evals/test/invocations.test.ts +++ b/evals/test/invocations.test.ts @@ -5,11 +5,7 @@ const record = (exitCode: number, args: ReadonlyArray): string => `${JSON.stringify({ exitCode, args })}\n`; describe("parseInvocationLog", () => { - it("returns an empty array for an empty log", () => { - expect(parseInvocationLog("")).toEqual([]); - }); - - it("parses each JSON line, keeping a spaced argument as one field", () => { + it("parses one invocation per JSON line", () => { const log = record(0, ["docs", "search", "content type"]) + record(1, ["mapi", "types", "--envId", "x", "--mapiKey", ""]); @@ -31,14 +27,6 @@ describe("parseInvocationLog", () => { expect(parseInvocationLog(log)).toEqual([{ exitCode: 0, args: ["docs"] }]); }); - - it("keeps an empty-string argument", () => { - const log = record(0, ["mapi", "types", "--envId", ""]); - - expect(parseInvocationLog(log)).toEqual([ - { exitCode: 0, args: ["mapi", "types", "--envId", ""] }, - ]); - }); }); describe("countInvocations", () => { diff --git a/evals/test/policy.test.ts b/evals/test/policy.test.ts index ec4f149..37e5758 100644 --- a/evals/test/policy.test.ts +++ b/evals/test/policy.test.ts @@ -1,20 +1,21 @@ import { describe, expect, it } from "vitest"; -import { none, some } from "../../src/lib/option.js"; -import { err, isErr, isOk, ok } from "../../src/lib/result.js"; -import { - AGENT_PERMISSION_RULES, - applyToolPolicy, - checkWebFetch, - DENIAL_PREFIX, - findPathOutsideWorkspace, - type ToolPolicy, - workspaceDirAliases, -} from "../lib/policy.js"; +import { err, ok } from "../../src/lib/result.js"; +import { applyToolPolicy, DENIAL_PREFIX, type ToolPolicy } from "../lib/policy.js"; const workspace = "/var/folders/ab/kontent-eval-x"; -const workspaceDirs = workspaceDirAliases(workspace); +const key = "secret-key-123"; +const policy: ToolPolicy = { workspaceDir: workspace, mapiKey: key }; -describe("findPathOutsideWorkspace", () => { +const bash = (command: string, activePolicy: ToolPolicy = policy) => + applyToolPolicy(activePolicy, { toolName: "Bash", input: { command } }); + +const webFetch = (url: string, activePolicy: ToolPolicy = policy) => + applyToolPolicy(activePolicy, { toolName: "WebFetch", input: { url, prompt: "how do I do x" } }); + +const outsideWorkspace = (token: string) => + err(`${DENIAL_PREFIX}command references a path outside the workspace: ${token}`); + +describe("Bash", () => { it.each([ "kontent mapi GET /types", "kontent mapi GET types --mapiKey $EVALS_MAPI_KEY", @@ -25,7 +26,7 @@ describe("findPathOutsideWorkspace", () => { "cat x > /dev/null", "echo a|grep b", ])("allows %s", (command) => { - expect(findPathOutsideWorkspace(command, workspaceDirs)).toEqual(none); + expect(bash(command)).toEqual(ok(undefined)); }); it.each([ @@ -35,8 +36,6 @@ describe("findPathOutsideWorkspace", () => { ["cat /Users/someone/.zshrc", "/Users/someone/.zshrc"], [`cat ${workspace}/../other/secret`, `${workspace}/../other/secret`], ["cat ../secret", "../secret"], - ["cd ..", ".."], - ["cd foo/..", "foo/.."], ["cat /var/folders/ab/other-dir/file", "/var/folders/ab/other-dir/file"], ['grep -r key "/Users/someone/src"', '"/Users/someone/src"'], ["ls /etc", "/etc"], @@ -46,27 +45,30 @@ describe("findPathOutsideWorkspace", () => { ["true;cat /etc/passwd", "/etc/passwd"], ["(cat /etc/passwd)", "/etc/passwd"], ])("denies %s", (command, expected) => { - expect(findPathOutsideWorkspace(command, workspaceDirs)).toEqual(some(expected)); + expect(bash(command)).toEqual(outsideWorkspace(expected)); }); -}); -describe("workspaceDirAliases", () => { - it("pairs the mkdtemp path with its /private twin", () => { - expect(workspaceDirAliases("/var/x")).toEqual(["/var/x", "/private/var/x"]); - expect(workspaceDirAliases("/private/var/x")).toEqual(["/private/var/x", "/var/x"]); + it("treats a /private-prefixed workspace and its mkdtemp spelling as the same dir", () => { + const privatePolicy: ToolPolicy = { ...policy, workspaceDir: `/private${workspace}` }; + + expect(bash(`cat ${workspace}/body.json`, privatePolicy)).toEqual(ok(undefined)); + expect(bash(`cat /private${workspace}/body.json`, privatePolicy)).toEqual(ok(undefined)); }); -}); -describe("checkWebFetch", () => { - const key = "secret-key-123"; - const policy: ToolPolicy = { workspaceDir: workspace, mapiKey: key }; + it("denies malformed input", () => { + const verdict = applyToolPolicy(policy, { toolName: "Bash", input: { notCommand: "oops" } }); + + expect(verdict).toEqual(err(`${DENIAL_PREFIX}malformed Bash input`)); + }); +}); +describe("WebFetch", () => { it.each([ "https://kontent.ai/learn/docs/apis/openapi/management-api-v2", "https://KONTENT.AI/learn", "http://kontent.ai/", ])("allows %s", (url) => { - expect(isOk(checkWebFetch(policy, url))).toBe(true); + expect(webFetch(url)).toEqual(ok(undefined)); }); it.each([ @@ -77,78 +79,28 @@ describe("checkWebFetch", () => { [`https://kontent.ai/?k=${key}`, "url contains the Management API key"], [`https://example.com/?k=${key}`, "url contains the Management API key"], ])("denies %s", (url, expected) => { - expect(checkWebFetch(policy, url)).toEqual(err(expected)); + expect(webFetch(url)).toEqual(err(`${DENIAL_PREFIX}${expected}`)); }); it("does not treat an empty key as contained in every url", () => { - const noKeyPolicy: ToolPolicy = { workspaceDir: workspace, mapiKey: "" }; - expect(checkWebFetch(noKeyPolicy, "https://kontent.ai/")).toEqual(ok(undefined)); - }); -}); - -describe("applyToolPolicy", () => { - const policy: ToolPolicy = { workspaceDir: workspace, mapiKey: "secret" }; + const noKeyPolicy: ToolPolicy = { ...policy, mapiKey: "" }; - it("allows a Bash call inside the workspace", () => { - const verdict = applyToolPolicy(policy, { - toolName: "Bash", - input: { command: "kontent auth status" }, - }); - - expect(verdict).toEqual(ok(undefined)); - }); - - it("denies a Bash call outside the workspace, prefixed with DENIAL_PREFIX", () => { - const verdict = applyToolPolicy(policy, { - toolName: "Bash", - input: { command: "cat /etc/passwd" }, - }); - - expect(isErr(verdict)).toBe(true); - expect(isErr(verdict) && verdict.error).toBe( - `${DENIAL_PREFIX}command references a path outside the workspace: /etc/passwd`, - ); - }); - - it("allows a WebFetch call to kontent.ai", () => { - const verdict = applyToolPolicy(policy, { - toolName: "WebFetch", - input: { url: "https://kontent.ai/learn", prompt: "how do I do x" }, - }); - - expect(verdict).toEqual(ok(undefined)); - }); - - it("denies a WebFetch call to another host", () => { - const verdict = applyToolPolicy(policy, { - toolName: "WebFetch", - input: { url: "https://example.com/", prompt: "how do I do x" }, - }); - - expect(verdict).toEqual(err(`${DENIAL_PREFIX}host is not allowed: example.com`)); + expect(webFetch("https://kontent.ai/", noKeyPolicy)).toEqual(ok(undefined)); }); it("denies malformed input", () => { - const verdict = applyToolPolicy(policy, { - toolName: "Bash", - input: { notCommand: "oops" }, - }); + const verdict = applyToolPolicy(policy, { toolName: "WebFetch", input: { url: 1 } }); - expect(verdict).toEqual(err(`${DENIAL_PREFIX}malformed Bash input`)); + expect(verdict).toEqual(err(`${DENIAL_PREFIX}malformed WebFetch input`)); }); +}); - it("allows any other tool", () => { - const verdict = applyToolPolicy(policy, { - toolName: "Read", - input: { path: "/etc/passwd" }, - }); +describe("other tools", () => { + // The hook only inspects Bash and WebFetch; every other tool is kept off the + // agent by the `tools` list in agent.ts, so the policy itself stays open. + it("leaves any other tool to the agent's tool list", () => { + const verdict = applyToolPolicy(policy, { toolName: "Read", input: { path: "README.md" } }); expect(verdict).toEqual(ok(undefined)); }); }); - -describe("AGENT_PERMISSION_RULES", () => { - it("scopes WebFetch to the allowed hosts rather than allowing it bare", () => { - expect(AGENT_PERMISSION_RULES).toEqual(["Bash", "WebFetch(domain:kontent.ai)"]); - }); -}); diff --git a/evals/test/prompt.test.ts b/evals/test/prompt.test.ts index 0debae5..72fdecd 100644 --- a/evals/test/prompt.test.ts +++ b/evals/test/prompt.test.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from "vitest"; -import { buildPreamble, buildTaskPrompt } from "../lib/prompt.js"; +import { buildTaskPrompt } from "../lib/prompt.js"; describe("buildTaskPrompt", () => { it("interpolates the env id and workspace dir, and appends the task body after a blank line", () => { @@ -11,13 +11,6 @@ describe("buildTaskPrompt", () => { expect(prompt).toContain("env-123"); expect(prompt).toContain("/tmp/workspace-abc"); - expect(prompt).not.toContain("{{"); expect(prompt.endsWith("\n\nDo the thing.")).toBe(true); }); }); - -describe("buildPreamble", () => { - it("renders exactly six lines", () => { - expect(buildPreamble("env-123", "/tmp/workspace-abc").split("\n")).toHaveLength(6); - }); -}); diff --git a/evals/test/report.test.ts b/evals/test/report.test.ts index 871ac56..dbb4190 100644 --- a/evals/test/report.test.ts +++ b/evals/test/report.test.ts @@ -94,137 +94,75 @@ const assertions: ReadonlyArray = [ ]; describe("renderTaskReport", () => { - it("includes the verdict heading, stats, assertions, tool calls and denials", () => { + it("renders a failed task", async () => { const report = renderTaskReport("content-type-with-snippet", trace, "FAIL", assertions); - expect(report).toContain("# content-type-with-snippet: FAIL"); - expect(report).toContain("turns 2"); - expect(report).toContain("failed 1 | denied 1"); - expect(report).toContain("cost $0.42"); - expect(report).toContain("tokens in 1000 | out 200 | cache 350"); - expect(report).toContain("- PASS article-type-exists content types: Article"); - expect(report).toContain("- FAIL seo-snippet-exists snippets: none"); - expect(report).toContain("`kontent content-type list`"); - expect(report).toContain("Content type not found"); - expect(report).toContain(' > agent: "Trying again."'); - expect(report).toContain("## Agent final reply\nDone."); + await expect(report).toMatchFileSnapshot("./__snapshots__/taskReport.fail.md"); }); - it("lists a denied call under ## Denied", () => { - const report = renderTaskReport("t", trace, "FAIL", assertions); - const deniedSection = report.split("## Denied")[1] ?? ""; - - expect(deniedSection).toContain( - `- Bash \`cat /etc/passwd\` (${DENIAL_PREFIX}outside workspace)`, - ); - }); - - it("renders a WebFetch call with its url, prompt and timing", () => { - const report = renderTaskReport("t", trace, "FAIL", assertions); - - expect(report).toContain( - '5. ok 2.4s WebFetch https://kontent.ai/learn/docs/apis/openapi/management-api-v2 ("taxonomy term shape")', - ); - expect(report).toContain("docs 0 | fetch 1"); - }); - - it("labels calls ok, FAIL, DENIED and NORESULT by outcome", () => { - const report = renderTaskReport("t", trace, "FAIL", assertions); - - expect(report).toContain("1. ok"); - expect(report).toContain("2. FAIL"); - expect(report).toContain("3. DENIED"); - expect(report).toContain("4. NORESULT"); - }); - - it("renders agent text before the first call as a blockquote under ## Tool calls", () => { - const report = renderTaskReport("t", trace, "FAIL", assertions); - const toolCallsSection = report.split("## Tool calls")[1] ?? ""; - - expect(toolCallsSection.trimStart().startsWith('> agent: "Let me look around first."')).toBe( - true, - ); - }); - - it("omits an ## Error section when no error message is given", () => { - const report = renderTaskReport("t", trace, "FAIL", assertions); - - expect(report).not.toContain("## Error"); - }); - - it("renders an ## Error section with the message when given", () => { + it("renders an errored task", async () => { const report = renderTaskReport("t", trace, "ERROR", [], "the check threw: boom"); - expect(report).toContain("## Error"); - expect(report).toContain("the check threw: boom"); + await expect(report).toMatchFileSnapshot("./__snapshots__/taskReport.error.md"); }); }); describe("renderRunReport", () => { - it("includes the summary table, totals and friction sections", () => { - const summary: RunSummary = { - header: { - model: "opus", - tools: ["Bash", "WebFetch"], - permissionRules: ["Bash", "WebFetch(domain:kontent.ai)"], - maxTurns: 40, - cliVersion: "0.9.2", - gitSha: "abc1234", - envId: "env-1", - preambleHash: "deadbeef1234", - startedAt: "2026-09-10T00:00:00.000Z", - finishedAt: "2026-09-10T00:10:00.000Z", - }, - rows: [ - { - id: "content-type-with-snippet", - verdict: "FAIL", - turns: 2, - toolCallCount: 2, - cliInvocationCount: 2, - failedCliInvocationCount: 1, - deniedCallCount: 1, - helpLookupCount: 0, - docsLookupCount: 0, - webFetchCount: 1, - costUsd: 0.42, - inputTokens: 12_000, - outputTokens: 3000, - durationMs: 90_000, - stopReason: "completed", - }, - ], - }; - - const report = renderRunReport(summary, [ + const summary: RunSummary = { + header: { + model: "opus", + tools: ["Bash", "WebFetch"], + permissionRules: ["Bash", "WebFetch(domain:kontent.ai)"], + maxTurns: 40, + cliVersion: "0.9.2", + gitSha: "abc1234", + envId: "env-1", + preambleHash: "deadbeef1234", + startedAt: "2026-09-10T00:00:00.000Z", + finishedAt: "2026-09-10T00:10:00.000Z", + }, + rows: [ { id: "content-type-with-snippet", - trace: { ...trace, toolCalls: [...trace.toolCalls, heredocFailedCall] }, + verdict: "FAIL", + turns: 2, + toolCallCount: 2, + cliInvocationCount: 2, + failedCliInvocationCount: 1, + deniedCallCount: 1, + helpLookupCount: 0, + docsLookupCount: 0, + webFetchCount: 1, + costUsd: 0.42, + inputTokens: 12_000, + outputTokens: 3000, + durationMs: 90_000, + stopReason: "completed", }, - ]); - - expect(report).toContain("# Eval run: opus @ env-1"); - expect(report).toContain("rules Bash, WebFetch(domain:kontent.ai)"); - expect(report).toContain("max turns 40"); - expect(report).toContain( - "| task | verdict | turns | calls | cli | failed | denied | help | docs | fetch | cost | in | out | time |", - ); - expect(report).toContain( - "| content-type-with-snippet | FAIL | 2 | 2 | 2 | 1 | 1 | 0 | 0 | 1 | $0.42 | 12k | 3k |", - ); - expect(report).toContain("### Web fetches"); - expect(report).toContain( - '- content-type-with-snippet: ok WebFetch https://kontent.ai/learn/docs/apis/openapi/management-api-v2 ("taxonomy term shape")', - ); - expect(report).toContain("0/1 passed"); - expect(report).toContain("## Friction"); - expect(report).toContain("### Failed calls"); - expect(report).toContain("`kontent content-type get --codename missing`"); - expect(report).toContain("Content type not found"); + ], + }; + + const traces = [ + { + id: "content-type-with-snippet", + trace: { ...trace, toolCalls: [...trace.toolCalls, heredocFailedCall] }, + }, + ]; + + it("renders a run", async () => { + const report = renderRunReport(summary, traces); + + await expect(report).toMatchFileSnapshot("./__snapshots__/runReport.md"); + }); + + it("lists only the first line of a failed heredoc command", () => { + const report = renderRunReport(summary, traces); + expect(report).toContain("`cat > body.json <<'EOF'` Error: HTTP 400 Bad Request"); expect(report).not.toContain('{"name":"x"}'); - expect(report).toContain("### content-type-with-snippet"); - expect(report).toContain("Done."); - expect(report).not.toContain("Most help lookups"); + }); + + it("omits the help lookup section when no task looked up help", () => { + expect(renderRunReport(summary, traces)).not.toContain("Most help lookups"); }); }); diff --git a/evals/test/results.test.ts b/evals/test/results.test.ts index d0cfbfd..7f38fea 100644 --- a/evals/test/results.test.ts +++ b/evals/test/results.test.ts @@ -42,7 +42,7 @@ describe("writeTaskResult redaction", () => { expect(markdown).not.toContain(mapiKey); }); - it("leaves the written task files untouched for an empty secret", async () => { + it("does not mangle the written task files when the secret is empty", async () => { const runDir = await mkdtemp(join(tmpdir(), "kontent-eval-results-")); await mkdir(join(runDir, "tasks")); const errorMessage = "nothing sensitive here"; diff --git a/evals/test/toolCalls.test.ts b/evals/test/toolCalls.test.ts index e85ad4e..f9ba72c 100644 --- a/evals/test/toolCalls.test.ts +++ b/evals/test/toolCalls.test.ts @@ -65,21 +65,7 @@ describe("collectToolCalls", () => { expect(calls[0]?.stdout).toBe(""); }); - it("classifies a call listed only in permission_denials as denied", () => { - const calls = collectToolCalls([ - createAssistantWebFetch("t1", "https://example.com/", "what is the request body shape"), - createResultMessage({ - permissionDenials: [ - { tool_name: "WebFetch", tool_use_id: "t1", tool_input: { url: "https://example.com/" } }, - ], - }), - ]); - - expect(calls[0]?.outcome).toBe("denied"); - expect(calls[0]?.stderr).toBe("not pre-approved by the permission rules"); - }); - - it("matches denials by tool_use_id, not by url: an earlier ok fetch to the same url stays ok", () => { + it("classifies a call listed only in permission_denials as denied, matched by tool_use_id", () => { const calls = collectToolCalls([ createAssistantWebFetch("t1", "https://kontent.ai/learn", "what is the request body shape"), createWebFetchResult("t1", { result: "first" }), @@ -95,8 +81,10 @@ describe("collectToolCalls", () => { }), ]); + // The earlier fetch to the same url stays ok: denials are keyed by id, not by url. expect(calls[0]?.outcome).toBe("ok"); expect(calls[1]?.outcome).toBe("denied"); + expect(calls[1]?.stderr).toBe("not pre-approved by the permission rules"); }); it("classifies a failed command from is_error", () => { @@ -131,13 +119,4 @@ describe("collectToolCalls", () => { expect(calls[0]?.stdout).toBe("first output"); expect(calls[1]?.stdout).toBe("second output"); }); - - it("classifies a plain successful call as ok", () => { - const calls = collectToolCalls([ - createAssistantBash("t1", "kontent content-type list"), - createBashResult("t1", { stdout: "Article\n", stderr: "" }), - ]); - - expect(calls[0]?.outcome).toBe("ok"); - }); }); diff --git a/evals/test/transcript.test.ts b/evals/test/transcript.test.ts index ba8cf2c..a2be728 100644 --- a/evals/test/transcript.test.ts +++ b/evals/test/transcript.test.ts @@ -20,7 +20,7 @@ const runWith = (messages: ReadonlyArray, extra: Partial = }); describe("buildTaskTrace", () => { - it("counts web fetches across a run", () => { + it("counts web fetches and denied calls across a run", () => { const trace = buildTaskTrace( runWith([ createAssistantWebFetch( @@ -31,11 +31,14 @@ describe("buildTaskTrace", () => { createWebFetchResult("t4", { result: "..." }), createAssistantBash("t5", "kontent content-type list"), createBashResult("t5", { stdout: "", stderr: "" }), + createAssistantBash("t6", "cat ~/.ssh/id_rsa"), + createDeniedResult("t6", "command references a path outside the workspace: ~/.ssh/id_rsa"), ]), [], ); expect(trace.webFetchCount).toBe(1); + expect(trace.deniedCallCount).toBe(1); }); it("derives invocation counts from the shim's invocation log, not the transcript", () => { @@ -43,8 +46,15 @@ describe("buildTaskTrace", () => { { exitCode: 0, args: ["docs", "search", "taxonomy"] }, { exitCode: 1, args: ["mapi", "--help"] }, ]; + // Would double every count if the transcript were consulted. + const messages = [ + createAssistantBash("t1", "kontent docs search taxonomy"), + createBashResult("t1", { stdout: "", stderr: "" }), + createAssistantBash("t2", "kontent mapi --help"), + createBashResult("t2", { stdout: "", stderr: "", isError: true }), + ]; - const trace = buildTaskTrace(runWith([]), invocations); + const trace = buildTaskTrace(runWith(messages), invocations); expect(trace.invocations).toBe(invocations); expect(trace.cliInvocationCount).toBe(2); @@ -101,20 +111,6 @@ describe("buildTaskTrace", () => { expect(trace.finalReply).toBe("The actual final answer."); }); - it("reads turns, cost and duration off the terminal result message", () => { - const trace = buildTaskTrace( - runWith([ - createAssistantText("Working on it."), - createResultMessage({ numTurns: 3, durationMs: 4200, costUsd: 0.12 }), - ]), - [], - ); - - expect(trace.numbers.turns).toBe(3); - expect(trace.numbers.durationMs).toBe(4200); - expect(trace.numbers.costUsd).toBe(0.12); - }); - it("falls back to usage fields when modelUsage is empty", () => { const trace = buildTaskTrace( runWith([ @@ -177,24 +173,4 @@ describe("buildTaskTrace", () => { cacheCreationTokens: 6, }); }); - - it("counts denied calls across a run", () => { - const trace = buildTaskTrace( - runWith([ - createAssistantBash("t1", "kontent content-type list"), - createBashResult("t1", { stdout: "", stderr: "" }), - createAssistantBash("t3", "cat ~/.ssh/id_rsa"), - createDeniedResult("t3", "command references a path outside the workspace: ~/.ssh/id_rsa"), - ]), - [], - ); - - expect(trace.deniedCallCount).toBe(1); - }); - - it("carries stopReason straight through from the run", () => { - const trace = buildTaskTrace(runWith([], { stopReason: "timeout" }), []); - - expect(trace.stopReason).toBe("timeout"); - }); }); From b3c5f25ff78df95ea6e2d4c74ce2e84b98309452 Mon Sep 17 00:00:00 2001 From: Ivan Kiral Date: Tue, 22 Sep 2026 15:51:53 +0200 Subject: [PATCH 2/2] test: swap redundant mapi command cases for transport failure and stdin input --- test/integration/mapi.test.ts | 2 +- test/integration/mapiCommand.test.ts | 52 ++++++++++++++++------------ 2 files changed, 31 insertions(+), 23 deletions(-) diff --git a/test/integration/mapi.test.ts b/test/integration/mapi.test.ts index f4e9796..77fb474 100644 --- a/test/integration/mapi.test.ts +++ b/test/integration/mapi.test.ts @@ -83,7 +83,7 @@ describe("performRawMapiRequest", () => { expect(contentTypes.map((header) => header.value)).toEqual(["text/plain"]); }); - it("adds no Authorization of its own when the client has no token", async () => { + it("sends a caller-supplied Authorization header as the only one", async () => { const { requests } = await run([typesRoute], { token: undefined, params: { headers: [{ name: "Authorization", value: "Bearer caller-token" }] }, diff --git a/test/integration/mapiCommand.test.ts b/test/integration/mapiCommand.test.ts index 8404689..54ed6a7 100644 --- a/test/integration/mapiCommand.test.ts +++ b/test/integration/mapiCommand.test.ts @@ -5,7 +5,7 @@ import { join } from "node:path"; import { promisify } from "node:util"; import { afterAll, beforeAll, beforeEach, describe, expect, it, vi } from "vitest"; import yargs from "yargs"; -import { register } from "../../src/commands/mapi/request.js"; +import { register as registerMapiCommand } from "../../src/commands/mapi/request.js"; import type { MapiRequestParams } from "../../src/core/mapi/request.js"; import { performRawMapiRequest } from "../../src/core/mapi/request.js"; import { getValidAccessToken } from "../../src/lib/auth/tokenAccess.js"; @@ -31,7 +31,7 @@ type CommandRun = Readonly<{ failure: string | undefined; stdout: string; stderr // Both streams are always captured: every 2xx writes somewhere, and a test that // only cares about the parsed arguments must not spill that into the runner's output. const runCommand = async (argv: ReadonlyArray): Promise => { - const parser = register( + const parser = registerMapiCommand( yargs([...argv]) .strict() .exitProcess(false) @@ -94,13 +94,6 @@ describe("kontent mapi argument handling", () => { expect(lastParams().headers).toContainEqual({ name: "X-Foo", value: "1" }); }); - it("accepts -H after the endpoint too", async () => { - const { failure } = await runCommand(["types", "-H", "X-Foo: 1", "--envId", ENV_ID]); - - expect(failure).toBeUndefined(); - expect(lastParams().endpoint).toBe("types"); - }); - it("collects a repeated -H into one header list", async () => { await runCommand(["-H", "X-Foo: 1", "-H", "X-Bar: 2", "types", "--envId", ENV_ID]); @@ -165,6 +158,17 @@ describe("kontent mapi argument handling", () => { expect(process.exitCode).toBe(1); }); + it("reports a transport failure on stderr and fails", async () => { + vi.mocked(performRawMapiRequest).mockResolvedValueOnce( + err({ kind: "transport", message: "fetch failed: getaddrinfo ENOTFOUND" }), + ); + const { stdout, stderr } = await runCommand(["types", "--envId", ENV_ID]); + + expect(stdout).toBe(""); + expect(stderr).toContain("fetch failed: getaddrinfo ENOTFOUND"); + expect(process.exitCode).toBe(1); + }); + it("sends the file at --input as the request body", async () => { const path = join(tempDir, "body.json"); await writeFile(path, '{"name":"Article"}'); @@ -190,20 +194,24 @@ describe("kontent mapi argument handling", () => { expect(await lastParams().body?.text()).toBe('{"name":"Article"}'); }); - it.skipIf(process.platform === "win32")( - "sends an empty body for a character device with nothing in it", - async () => { - await runCommand(["types", "--input", "/dev/null", "--envId", ENV_ID]); - - expect(await lastParams().body?.text()).toBe(""); - expect(process.exitCode).toBeUndefined(); - }, - ); - - it("wires no abort of its own, leaving SIGINT to the telemetry handler", async () => { - await runCommand(["types", "--envId", ENV_ID]); + // Without nargs on --input, strict mode rejects the lone "-" as an unknown positional. + it("reads --input - from stdin, refusing when nothing is piped", async () => { + // The runner's stdin is not a terminal, so the flag is set by hand and removed after. + Object.defineProperty(process.stdin, "isTTY", { value: true, configurable: true }); + const { failure, stderr } = await runCommand([ + "types", + "--input", + "-", + "--envId", + ENV_ID, + ]).finally(() => { + delete (process.stdin as { isTTY?: boolean }).isTTY; + }); - expect(lastParams().abortSignal).toBeUndefined(); + expect(failure).toBeUndefined(); + expect(stderr).toContain("Nothing is piped to stdin"); + expect(process.exitCode).toBe(1); + expect(performRawMapiRequest).not.toHaveBeenCalled(); }); it("reports an unreadable --input file without calling the API", async () => {