diff --git a/.env.template b/.env.template index 2890ca1..c208c10 100644 --- a/.env.template +++ b/.env.template @@ -10,13 +10,11 @@ KONTENT_CLI_AMPLITUDE_API_KEY= # Base Kontent.ai domain; all endpoint URLs derive from it (app., manage.). -# Baked into the build from here; defaults to kontent.ai (production). Use -# devkontentmasters.com for dev. A runtime KONTENT_URL in your shell overrides it. +# Baked into the build from here; defaults to kontent.ai (production). KONTENT_URL=devkontentmasters.com # Auth0 tenant baked into the build; defaults here target the QA/dev tenant. -# A runtime KONTENT_AUTH0_* var in your shell overrides each. The requested -# scope is not configurable (fixed requirement of the CLI). +# A runtime KONTENT_AUTH0_* var in your shell overrides each. KONTENT_AUTH0_DOMAIN=login.devkontentmasters.com KONTENT_AUTH0_CLIENT_ID= KONTENT_AUTH0_AUDIENCE=https://app.kenticocloud.com/ @@ -25,7 +23,6 @@ KONTENT_AUTH0_AUDIENCE=https://app.kenticocloud.com/ # Telemetry opt-out (set to 1/true to disable). DO_NOT_TRACK is the standard equivalent. KONTENT_DO_NOT_TRACK= -DO_NOT_TRACK= # Verbose telemetry logging for debugging. KONTENT_TELEMETRY_DEBUG= @@ -43,3 +40,23 @@ E2E_SOURCE_ENV_ID= # Domain of the e2e project. The e2e run ignores KONTENT_URL above and defaults # to production kontent.ai; set this only when the test project lives elsewhere. E2E_KONTENT_URL=kontent.ai + +# --- Agent evals (pnpm evals:run; fails with an error when these are unset) --- +# Read from .env by vitest.evals.config.ts, same as the E2E_* block above. + +# Management API key of the dedicated evals project: access to all environments +# plus the Manage environments permission. +EVALS_MAPI_KEY= + +# Empty template environment that each eval run clones. Never written to. +EVALS_SOURCE_ENV_ID= + +# Model under test, passed to the Agent SDK. Defaults to "sonnet" when unset. +EVALS_MODEL= + +# Set to 1 to keep the cloned environment after the run instead of deleting it. +EVALS_KEEP_ENV= + +# Evals authenticate as your local Claude Code subscription login; the run +# fails fast if ANTHROPIC_API_KEY is set. Set this to 1 to run with a key anyway. +EVALS_ALLOW_API_KEY= diff --git a/.gitignore b/.gitignore index 869d6a3..6818279 100644 --- a/.gitignore +++ b/.gitignore @@ -13,5 +13,9 @@ my-kickstart/ .claude/worktrees/ .claude/plans/ +# Eval run reports; the folder itself stays tracked +evals/results/* +!evals/results/.gitkeep + # pnpm pack output *.tgz diff --git a/CLAUDE.md b/CLAUDE.md index 14112c1..b98eb40 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -23,15 +23,16 @@ Three layers, dependencies point downward only (`commands → core → lib`): - `src/core/**` — orchestration of business logic. Returns `Result`/`Option`; never writes to the console directly (logs only through a passed `Logger`). **Exception:** interactive commands may drive their own terminal UI from core — e.g. `src/core/project/bootstrap.ts` uses the prompts of `src/lib/ui/prompts.ts` (spinners, `confirm`/`select`, notes) directly because the flow is inherently interactive. Keep non-interactive core free of direct console writes. - `src/lib/**` — reusable primitives: `auth/`, `iapi/`, `mapi/`, `config/`, `telemetry/`, plus `result.ts` and `option.ts`. -Adding a command: export a `register: RegisterCommand` (see `src/commands/login/login.ts`), then add its import to the `register` array in the parent command or `src/commands/registry.ts`. Then run `pnpm docs:generate` (`scripts/generateCommandDocs.ts`) — it replays the registrations against a recording proxy and rewrites the generated docs: the marker-fenced command table in the root `README.md`, and the `` block in each command folder's `README.md` (created as a skeleton when missing). Prose outside the markers is handwritten — write command docs there, never inside the block. Two opt-out sets in the script: `commandsWithoutPage` (no colocated README) and `commandsWithoutIndexEntry` (no root-README table row; telemetry is there). The generator errors on a command-folder README with markers but no matching command (stale after rename/removal) — resolve by hand; it never deletes pages. +Adding a command: export a `register: RegisterCommand` (see `src/commands/login/login.ts`), then add its import to the `register` array in the parent command or `src/commands/registry.ts`. Then run `pnpm docs:generate` (`scripts/generateCommandDocs.ts`) — it replays the registrations against a recording proxy and rewrites the generated docs: the marker-fenced command table in the root `README.md`, and the `` block in each command's `README.md` (created as a skeleton when missing), placed in the leaf's own folder when it has one and in its parent group's folder otherwise. Prose outside the markers is handwritten — write command docs there, never inside the block. Two opt-out sets in the script: `commandsWithoutPage` (no colocated README) and `commandsWithoutIndexEntry` (no root-README table row; telemetry is there). The generator errors on a command-folder README with markers but no matching command (stale after rename/removal) — resolve by hand; it never deletes pages. ### API clients - `iapi` (`src/lib/iapi`) — internal Kontent.ai API; hand-rolled client, one file per endpoint, over `@kontent-ai/core-sdk`. Endpoint validators (the `schema` field) must be **`zod/mini`** (`import * as z from "zod/mini"`) — classic `zod` won't infer the payload. - `mapi` (`src/lib/mapi`) — public Management API via `@kontent-ai/management-sdk`. `src/lib/mapi/raw` is the deliberate opposite: a passthrough (no schema, no response interpretation) behind `kontent mapi`, where a 4xx/5xx is a result, not an error. It builds on core-sdk's `getDefaultHttpService` and turns the non-2xx it reports as errors back into results, reading the body off `error.details.adapterResponse`; retry, `Retry-After` and header merging are core-sdk's. Its doc comments carry the why: `raw/client.ts` for which SDK error reasons stay errors, `raw/contentType.ts` for the rule that decides whether a body is printed. -- `@kontent-ai/core-sdk` — shared HTTP/SDK layer both clients build on. +- `learn` (`src/lib/learn`) — tokenless `createFetchQuery` client over the Learn-MCP service (`https://learn-mcp.kontent.ai`, no auth, GET only) behind `kontent docs`. The `zod/mini` schemas declare only the fields the CLI reads; core-sdk hands back the raw payload, so unknown keys survive to stdout. `runtimeValidation.validateResponses` must stay on, otherwise the schema never runs. +- `@kontent-ai/core-sdk` — shared HTTP/SDK layer all clients build on. -**Commands build clients; core receives them.** The command builds the `iapiClient`/`mapiClient` and passes them into core (e.g. `performBootstrap(params, { logger, iapiClient, mapiClient })`); core never constructs clients itself. Auth failure is handled in the command, not surfaced as a core `Result` error. Same split for arguments: pure parsers live in `lib` (`mapi/raw/headers.ts`, `mapi/raw/method.ts`), reading what the invocation points at stays in the command, and each layer declares only the error kinds it raises. +**Commands build clients; core receives them.** The command builds the `iapiClient`/`mapiClient` and passes them into core (e.g. `performBootstrap(params, { logger, iapiClient, mapiClient })`); core never constructs clients itself. Auth failure is handled in the command, not surfaced as a core `Result` error. The `kontent login` token is the credential for both `iapi` and `mapi`: a logged-in user needs no separate API key, so only the environment id ever needs resolving. Same split for arguments: pure parsers live in `lib` (`mapi/raw/headers.ts`, `mapi/raw/method.ts`), reading what the invocation points at stays in the command, and each layer declares only the error kinds it raises. ### Output channels @@ -54,15 +55,17 @@ Core takes the logger as a parameter or inside its `deps` object; `createLoggerF - **ESM import extensions:** relative imports must end in `.js` (biome enforces `useImportExtensions`). - **No redundant wrappers.** Don't add a function that only forwards to another; reuse existing helpers instead of duplicating logic. - **Comments only for non-obvious "why".** No restating-the-code comments, no repeating a fact already stated elsewhere (put domain facts once, on the type), no justifying a change to the reviewer ("X already did Y, so..."). No emojis anywhere. -- **Exports first.** A module's exported types and functions go at the top, private helpers below them. Arrow consts are only called after module evaluation, so referring downward is safe. +- **Exports first.** A module's exported types and functions go at the top, private helpers below them. Below them, private helpers in call order, depth-first; a private constant sits directly above its one user. Arrow consts are only called after module evaluation, so referring downward is safe. - **No barrel files** except a deliberate public API. ## Testing -Vitest; `test/unit/` for pure unit tests, `test/integration/` for integration tests, `test/helpers/` for shared helpers. Command-level behavior (argument parsing, exit codes, which stream a message lands on) is tested by folding a command's `register` over a real yargs instance and faking only the core call underneath — see `test/integration/mapiCommand.test.ts`. Run `pnpm test`. Inject fakes into core instead of real I/O — for iapi reuse `test/helpers/iapiTestClient.ts` (real client over core-sdk's `HttpAdapter` seam, declarative routes). +Vitest; `test/unit/` for pure unit tests, `test/integration/` for integration tests, `test/helpers/` for shared helpers. Command-level behavior (argument parsing, exit codes, which stream a message lands on) is tested by folding a command's `register` over a real yargs instance and faking only the core call underneath — see `test/integration/mapiCommand.test.ts`. Run `pnpm test`. Inject fakes into core instead of real I/O — for iapi reuse `test/helpers/iapiTestClient.ts` (real client over core-sdk's `HttpAdapter` seam, declarative routes). Harness unit tests live in `evals/test/` (helpers in `evals/test/helpers/`) and run in `pnpm test`; only the paid agent run is opt-in. `test/e2e/` runs the built binary against a real Kontent.ai project (clone-per-run from an empty template env). Gated on `E2E_MAPI_KEY`/`E2E_SOURCE_ENV_ID` (fails fast with an error when unset). Run with `pnpm test:e2e` (own `vitest.e2e.config.ts`, loads `.env`); excluded from `pnpm test` and the before-halting gate. CI: `.github/workflows/e2e.yml` (master push, PRs, manual; fork PRs are skipped at the job level — no secret access). +`evals/` is a Vitest-driven agent-eval harness (`pnpm evals:run`, own `vitest.evals.config.ts`): the Agent SDK drives the built CLI with Bash (confined to a workspace dir) plus WebFetch (`kontent.ai` only, key-in-URL denied), against one cloned environment shared by every task, run sequentially in dependency order (a task whose parent did not PASS is recorded BLOCKED, no agent spawned). The policy is a pure reducer (`evals/lib/policy.ts`) wired into the `PreToolUse` hook by `evals/lib/agent.ts`, the only effectful module; tool calls and denials are derived from the run's raw messages in `evals/lib/toolCalls.ts`; CLI invocations and exit codes come from the shim's per-task log (`evals/lib/invocations.ts`), never from parsing shell text. The agent run never runs in `pnpm test` or CI, and is gated on `EVALS_MAPI_KEY`/`EVALS_SOURCE_ENV_ID` (which the CLI itself must never read). `evals/tasks//task.ts` (prompt + check together) grades the resulting state through the Management SDK; checks stay deterministic: return `Result` (a failed lookup is an `err`, never a failed assertion), and stay tolerant about names the task did not fix. Reports (`evals/lib/report/`, markdown only, no LLM calls) are a nice-to-have removable by deleting that folder and its one call site. Playbook: `evals/README.md`. + ## Telemetry Amplitude-based, see `TELEMETRY.md`. Env vars are read from `process.env` where they apply, never mapped onto yargs options — `src/index.ts` deliberately does not call `.env()`, so a stray `KONTENT_*` var cannot break an unrelated command. diff --git a/README.md b/README.md index 3713d16..7fd67f5 100644 --- a/README.md +++ b/README.md @@ -56,10 +56,13 @@ Each command supports `--help` for its own options. | Command | Description | | --- | --- | +| [`kontent docs search `](src/commands/docs/search/README.md) | Search Kontent.ai Learn docs and API reference, ranked by relevance | +| [`kontent docs endpoint `](src/commands/docs/endpoint/README.md) | Show matching API endpoints: method, URL, parameters, responses, code samples | +| [`kontent docs object `](src/commands/docs/object/README.md) | Show matching API reference objects and their properties | | [`kontent login`](src/commands/login/README.md) | Authenticate with Kontent.ai via Auth0 device flow | | [`kontent logout`](src/commands/logout/README.md) | Clear stored authentication tokens | | [`kontent mapi `](src/commands/mapi/README.md) | Send an authenticated request to the Management API | -| [`kontent project sample bootstrap`](src/commands/project/sample/README.md) | Clone a sample app for an environment and wire its .env | +| [`kontent project sample bootstrap`](src/commands/project/sample/bootstrap/README.md) | Clone a sample app for an environment and wire its .env | ## Global options diff --git a/evals/README.md b/evals/README.md new file mode 100644 index 0000000..8d4aefa --- /dev/null +++ b/evals/README.md @@ -0,0 +1,83 @@ +# Agent evals + +Measures how far an AI agent, acting like a fresh human user, gets when it drives the built +`kontent` CLI against a real Kontent.ai environment. One Agent SDK run per task; deterministic +checks in `evals/tasks//task.ts` grade the resulting environment state afterwards. This agent +run goes through Vitest, but not as part of `pnpm test` or CI - it costs money and needs a cloned +environment, so it is a deliberate opt-in. The harness's own unit tests (`evals/test/`) are plain +Vitest specs and run as part of `pnpm test` like any other unit test. + +## What the agent gets + +Two tools, declared in `evals/lib/policy.ts` and wired into the SDK options by `evals/lib/agent.ts`, +enforced by the `PreToolUse` hook via the pure policy reducer in `evals/lib/policy.ts`: + +- `Bash`, denied for any command referencing a path outside the task's workspace directory. This is + a file-access guard, not a sandbox: Bash has full network access, and the agent is trusted with it. +- `WebFetch`, allowed for `kontent.ai` only (the API reference lives at `kontent.ai/learn/...`), and + denied outright for a URL that contains the Management API key. The rule exists to measure, not to + contain: every fetch is a fallback the CLI's own docs did not cover. + +The preamble says nothing about the web or about `kontent docs`: which route the agent takes to the +API reference is part of what a run records (`docs` and `fetch` counts, and the report's "Web +fetches" section). Every fetch is a place where the CLI's own docs did not carry the agent. + +## Preconditions + +- `EVALS_MAPI_KEY` and `EVALS_SOURCE_ENV_ID` exported in your shell, or set in `.env`. +- Logged into Claude Code locally (`claude` on PATH, authenticated). +- `ANTHROPIC_API_KEY` unset. Evals authenticate as your local subscription login, not an API key - + see [Anthropic policy](#anthropic-policy) below. The run fails fast if it is set, unless you pass + `EVALS_ALLOW_API_KEY=1`. + +## Run it + +``` +EVALS_MODEL=sonnet pnpm evals:run +``` + +`globalSetup.ts` builds the CLI, clones `EVALS_SOURCE_ENV_ID`, and hands the environment id and the +built CLI's bin directory to the test file. `evals/run.eval.ts` then runs every task exported from +`evals/lib/registry.ts`, in dependency order, sequentially, against that one environment: if any of +a task's parents did not PASS, the task is recorded BLOCKED and no agent is spawned for it. Teardown +deletes the cloned environment unless you keep it (see below). + +### Knobs (environment variables) + +- `EVALS_MODEL` - model passed to the Agent SDK. Default `sonnet`. +- `EVALS_KEEP_ENV=1` - keep the cloned environment instead of deleting it at the end. +- `EVALS_ALLOW_API_KEY=1` - run even with `ANTHROPIC_API_KEY` set. + +## Output + +Each run writes to `evals/results/--/` (git-ignored; date and time are UTC, +taken at the start of the run): + +- `run.json` - header (model, tools, permission rules, max turns, CLI version, git sha, environment + id, preamble hash, timing) and one row per task (verdict, turns, tool call, failed, denied, help, + docs and fetch counts, cost, duration, stop reason). +- `report.md` - the same summary as a table, plus a friction section (every failed call, every + web fetch) and every task's final agent reply. Nice-to-have: + `evals/lib/report/` renders it from the same data as `run.json`, with no LLM calls; deleting that + folder and the one call into it from `evals/lib/results.ts` removes the feature cleanly. +- `tasks/.json` - the task's full trace (tool calls and fetches, agent text, numbers) plus its + verdict and assertions. +- `tasks/.raw.jsonl` - every raw Agent SDK message for that task, one per line. +- `tasks/.md` - the same task report rendered as markdown. +- `tasks/.invocations` - one JSON line per `kontent` process the agent ran (exit code and + argv, key redacted), the source of the `cli`, `failed`, `help` and `docs` columns. + +Each task also gets its own workspace directory under the OS temp dir (`kontent-eval--*`), +where the agent runs its commands. These are left in place after the run, not cleaned up, so they +stay available for inspection. + +## Adding a task + +1. `evals/tasks//task.ts` - export an `EvalTask` with an `id`, its `dependsOn` (the ids of its + parent tasks), a `prompt` phrased as a busy human would type it, with no hints about the CLI, and + a `check` that reads the environment through the Management SDK and returns assertions. A failed + lookup is an `err`, never a failed assertion. Codenames the prompt fixes are asserted exactly; + names are never asserted; content items are named only and found by name. Include a regression + assertion for anything an earlier task in the chain seeded. +2. Add its id to the `TaskId` union in `evals/lib/types.ts`. +3. Add it to `evals/lib/registry.ts`. diff --git a/evals/globalSetup.ts b/evals/globalSetup.ts new file mode 100644 index 0000000..82f83a5 --- /dev/null +++ b/evals/globalSetup.ts @@ -0,0 +1,114 @@ +import { execFile } from "node:child_process"; +import { chmod, mkdtemp, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { promisify } from "node:util"; +import type { TestProject } from "vitest/node"; +import { isErr } from "../src/lib/result.js"; +import { cloneTestEnvironment, deleteTestEnvironment } from "../test/helpers/environment.js"; +import { requireEvalsConfig } from "./lib/config.js"; +import { createRunDirectory } from "./lib/results.js"; + +declare module "vitest" { + interface ProvidedContext { + evals: Readonly<{ + envId: string; + cliBinDir: string; + runDir: string; + model: string; + mapiKey: string; + }>; + } +} + +const execFileAsync = promisify(execFile); +const repoRoot = fileURLToPath(new URL("..", import.meta.url)); + +// Runs once before evals/run.eval.ts: builds the CLI, clones the eval +// environment, and hands both to the test file via `inject("evals")`. +// Teardown deletes the clone unless EVALS_KEEP_ENV=1. +export const setup = async ({ provide }: TestProject): Promise<() => Promise> => { + failIfApiKeyIsSet(); + + const config = requireEvalsConfig(); + if (isErr(config)) { + throw new Error(config.error); + } + + process.stderr.write("Building the CLI...\n"); + await execFileAsync("pnpm", ["build"], { cwd: repoRoot }); + + const cliBinDir = await createCliShim(); + const model = readModel(); + const runDir = await createRunDirectory(join(repoRoot, "evals", "results"), model); + + process.stderr.write(`Cloning ${config.value.sourceEnvId}; this takes a while.\n`); + const environment = await cloneTestEnvironment(config.value, "evals"); + process.stderr.write(`Cloned as ${environment.name} (${environment.envId}).\n`); + + // Anything past this point that throws must still delete the clone: there + // is no returned teardown yet for Vitest to call on our behalf. + try { + provide("evals", { + envId: environment.envId, + cliBinDir, + runDir, + model, + mapiKey: config.value.mapiKey, + }); + } catch (cause) { + await deleteTestEnvironment(config.value, environment.envId); + throw cause; + } + + return async () => { + process.stderr.write(`Environment: ${environment.envId}. Run dir: ${runDir}\n`); + if (process.env.EVALS_KEEP_ENV === "1") { + process.stderr.write("EVALS_KEEP_ENV=1: leaving the environment in place.\n"); + return; + } + await deleteTestEnvironment(config.value, environment.envId); + process.stderr.write(`Deleted environment ${environment.envId}.\n`); + }; +}; + +// Decision: evals authenticate as the operator's own local Claude Code +// subscription login, never an API key, so the CLI's cost stays on the +// operator's plan rather than an org's Console budget. EVALS_ALLOW_API_KEY=1 +// is the deliberate escape hatch (e.g. a CI account that only has a key). +const failIfApiKeyIsSet = (): void => { + const hasApiKey = + process.env.ANTHROPIC_API_KEY !== undefined && process.env.ANTHROPIC_API_KEY !== ""; + if (hasApiKey && process.env.EVALS_ALLOW_API_KEY !== "1") { + throw new Error( + "ANTHROPIC_API_KEY is set. Evals run through your local Claude Code subscription login, not " + + "an API key. Unset it, or set EVALS_ALLOW_API_KEY=1 to run with it anyway.", + ); + } +}; + +const readModel = (): string => { + const model = process.env.EVALS_MODEL; + return model === undefined || model === "" ? "sonnet" : model; +}; + +// Creates a temp dir under the OS temp dir holding one executable bash script +// named `kontent`. evals/lib/agent.ts prepends this dir to the agent's PATH, +// so `kontent` resolves to the build `pnpm build` just produced above, never +// a stale global install. The script defers to evals/shim.ts, which runs the +// built CLI and, when $EVALS_INVOCATION_LOG is set, also writes the +// invocation log (see evals/lib/invocations.ts). Bash-only, so no Windows. +// The temp dir is not cleaned up. +const createCliShim = async (): Promise => { + const binDir = await mkdtemp(join(tmpdir(), "kontent-evals-bin-")); + const shimPath = join(binDir, "kontent"); + await writeFile( + shimPath, + `#!/usr/bin/env bash +exec node "${join(repoRoot, "evals", "shim.ts")}" "$@" +`, + ); + await chmod(shimPath, 0o755); + return binDir; +}; diff --git a/evals/lib/agent.ts b/evals/lib/agent.ts new file mode 100644 index 0000000..1df6bd4 --- /dev/null +++ b/evals/lib/agent.ts @@ -0,0 +1,156 @@ +import type { + HookCallback, + Options, + query as queryFn, + SDKMessage, +} from "@anthropic-ai/claude-agent-sdk"; +import { err, isErr, ok, type Result } from "../../src/lib/result.js"; +import { describeCause } from "./inspect.js"; +import { findResultMessage } from "./messages.js"; +import { AGENT_PERMISSION_RULES, AGENT_TOOLS, applyToolPolicy, type ToolPolicy } from "./policy.js"; + +export type AgentRun = Readonly<{ + messages: ReadonlyArray; + stderr: string; + stopReason: "completed" | "timeout" | "max_turns" | "error" | "not-run"; +}>; + +export type RunTaskAgentParams = Readonly<{ + prompt: string; + model: string; + workspaceDir: string; + envId: string; + mapiKey: string; + cliBinDir: string; + invocationLogPath: string; + timeoutMs: number; + maxTurns: number; +}>; + +// The SDK's own `query` function, injected rather than imported directly so a +// test can supply a fake stream without spawning the real Claude Code +// subprocess. +export type AgentSdkQuery = typeof queryFn; + +export const runTaskAgent = async ( + params: RunTaskAgentParams, + deps: Readonly<{ query: AgentSdkQuery }>, +): Promise> => { + const messages: SDKMessage[] = []; + const stderrChunks: string[] = []; + const abortController = new AbortController(); + const timeoutId = setTimeout(() => abortController.abort(), params.timeoutMs); + + try { + const stream = deps.query({ + prompt: params.prompt, + options: buildOptions(params, abortController, stderrChunks), + }); + + for await (const message of stream) { + messages.push(message); + } + + return ok({ + messages, + stderr: stderrChunks.join(""), + stopReason: deriveStopReason(messages), + }); + } catch (cause) { + if (abortController.signal.aborted) { + return ok({ messages, stderr: stderrChunks.join(""), stopReason: "timeout" }); + } + return err(`Running the agent failed: ${describeCause(cause)}`); + } finally { + clearTimeout(timeoutId); + } +}; + +const buildOptions = ( + params: RunTaskAgentParams, + abortController: AbortController, + stderrChunks: string[], +): Options => ({ + model: params.model, + // Restricts the built-in toolset itself, unlike `allowedTools`, which only + // auto-approves without narrowing what the model can attempt. + tools: [...AGENT_TOOLS], + allowedTools: [...AGENT_PERMISSION_RULES], + // "Don't prompt for permissions, deny if not pre-approved" + permissionMode: "dontAsk", + cwd: params.workspaceDir, + env: buildSubprocessEnv(params), + // Isolates the run from this repo's CLAUDE.md and the operator's own + // Claude Code settings/hooks, neither of which the agent under eval should see. + settingSources: [], + maxTurns: params.maxTurns, + abortController, + persistSession: false, + stderr: (data) => stderrChunks.push(data), + hooks: { + PreToolUse: [ + { + hooks: [ + createToolGuardHook({ + workspaceDir: params.workspaceDir, + mapiKey: params.mapiKey, + }), + ], + }, + ], + }, +}); + +// An explicit allowlist, never a spread of process.env, so the harness' own +// credentials (KONTENT_*, E2E_*, EVALS_SOURCE_ENV_ID, ANTHROPIC_*) never +// reach the agent under eval. +const buildSubprocessEnv = (params: RunTaskAgentParams): Record => ({ + PATH: `${params.cliBinDir}:${process.env.PATH ?? ""}`, + HOME: process.env.HOME, + // Keys the macOS Keychain lookup for the stored login. + USER: process.env.USER, + // Where the CLI and node put temp files. + TMPDIR: process.env.TMPDIR, + // Locale; keeps tool output UTF-8. + LANG: process.env.LANG, + // Terminal type; chalk and the claude binary read it. + TERM: process.env.TERM, + EVALS_MAPI_KEY: params.mapiKey, + // Read by the `kontent` shim to append its per-invocation record; see + // evals/globalSetup.ts and evals/lib/invocations.ts. + EVALS_INVOCATION_LOG: params.invocationLogPath, +}); + +const createToolGuardHook = + (policy: ToolPolicy): HookCallback => + async (input) => + input.hook_event_name === "PreToolUse" + ? toHookOutput( + applyToolPolicy(policy, { toolName: input.tool_name, input: input.tool_input }), + ) + : {}; + +const toHookOutput = (verdict: Result): Awaited> => + isErr(verdict) + ? { + hookSpecificOutput: { + hookEventName: "PreToolUse", + permissionDecision: "deny", + permissionDecisionReason: verdict.error, + }, + } + : {}; + +const deriveStopReason = (messages: ReadonlyArray): AgentRun["stopReason"] => { + const result = findResultMessage(messages); + if (result === undefined) { + return "error"; + } + if (result.subtype === "success") { + return "completed"; + } + if (result.subtype === "error_max_turns") { + return "max_turns"; + } + return "error"; +}; diff --git a/evals/lib/codenames.ts b/evals/lib/codenames.ts new file mode 100644 index 0000000..981777f --- /dev/null +++ b/evals/lib/codenames.ts @@ -0,0 +1,16 @@ +// The default workflow, collection and language of every environment share this codename. +export const DEFAULT_CODENAME = "default"; +export const PUBLISHED_STEP_CODENAME = "published"; +export const SCHEDULED_STEP_CODENAME = "scheduled"; + +export const articleElementCodenames = [ + "title", + "body", + "author", + "summary", + "url_slug", + "hero_image", + "related", + "topics", + "publish_date", +] as const; diff --git a/evals/lib/config.ts b/evals/lib/config.ts new file mode 100644 index 0000000..ebbbdd1 --- /dev/null +++ b/evals/lib/config.ts @@ -0,0 +1,23 @@ +import { err, isErr, ok, type Result } from "../../src/lib/result.js"; +import { readRequiredEnvVars } from "../../test/helpers/requiredEnv.js"; + +export type EvalsConfig = Readonly<{ + mapiKey: string; + sourceEnvId: string; +}>; + +// The EVALS_* names stay distinct from the CLI's KONTENT_* variables and from +// the e2e suite's E2E_* ones, so an eval run can never touch either project. +export const requireEvalsConfig = (): Result => { + const vars = readRequiredEnvVars(["EVALS_MAPI_KEY", "EVALS_SOURCE_ENV_ID"]); + + if (isErr(vars)) { + return err(describeMissing(vars.error)); + } + + return ok({ mapiKey: vars.value.EVALS_MAPI_KEY, sourceEnvId: vars.value.EVALS_SOURCE_ENV_ID }); +}; + +const describeMissing = (missing: ReadonlyArray): string => + `Missing eval environment variables: ${missing.join(", ")}. ` + + "Export them in the shell that launches the eval run, or set them in .env."; diff --git a/evals/lib/inspect.ts b/evals/lib/inspect.ts new file mode 100644 index 0000000..afe8d8f --- /dev/null +++ b/evals/lib/inspect.ts @@ -0,0 +1,133 @@ +import type { + ContentItemModels, + ContentTypeElements, + LanguageVariantModels, + TaxonomyModels, + WorkflowModels, +} from "@kontent-ai/management-sdk"; +import { none, type Option, some } from "../../src/lib/option.js"; +import { err, ok, type Result, tryAsync } from "../../src/lib/result.js"; +import { isNotFoundError } from "../../test/helpers/environment.js"; +import type { Assertion } from "./types.js"; + +export const lookup = (what: string, run: () => Promise): Promise> => + tryAsync(run, (cause) => `${what} failed: ${describeCause(cause)}`); + +// A 404 is a grading outcome (the agent did not create the thing), every other +// failure is the harness' problem. +export const lookupByCodename = async ( + what: string, + run: () => Promise, +): Promise, string>> => { + try { + return ok(some(await run())); + } catch (cause) { + return isNotFoundError(cause) ? ok(none) : err(`${what} failed: ${describeCause(cause)}`); + } +}; + +export const describeCause = (cause: unknown): string => { + if (cause instanceof Error) { + return cause.message; + } + + // The management SDK throws ContentManagementBaseKontentError, a plain + // class that does not extend Error, so String(cause) would yield "[object Object]". + return hasStringMessage(cause) ? cause.message : String(cause); +}; + +export const describeList = (values: ReadonlyArray): string => + values.length === 0 ? "none" : values.join(", "); + +export const missingAssertion = (id: string, what: string): Assertion => ({ + id, + passed: false, + evidence: `${what} not found`, +}); + +export const findElementByCodename = ( + elements: ReadonlyArray, + codename: string, +): ContentTypeElements.ContentTypeElementModel | undefined => + elements.find((element) => element.codename === codename); + +// The Management API prefixes snippet element codenames with the snippet codename +// (meta_title is stored as seo__meta_title), so both spellings count. +export const findSnippetElementByCodename = ( + snippet: Readonly<{ + codename: string; + elements: ReadonlyArray; + }>, + codename: string, +): ContentTypeElements.ContentTypeElementModel | undefined => + snippet.elements.find( + (element) => + element.codename === codename || element.codename === `${snippet.codename}__${codename}`, + ); + +export const findItemsByName = ( + items: ReadonlyArray, + name: string, +): ReadonlyArray => + items.filter((item) => item.name.toLowerCase().includes(name.toLowerCase())); + +export const truncate = (text: string, maxLength = 200): string => + text.length > maxLength ? `${text.slice(0, maxLength)}...` : text; + +export const flattenTaxonomyTerms = ( + terms: ReadonlyArray, +): ReadonlyArray => + terms.flatMap((term) => [term, ...flattenTaxonomyTerms(term.terms)]); + +export const isVariantInStep = ( + variant: LanguageVariantModels.ContentItemLanguageVariant, + workflows: ReadonlyArray, + stepCodename: string, +): boolean => { + const found = findVariantStep(variant, workflows); + return found !== undefined && found.step.codename === stepCodename; +}; + +export const describeVariantStep = ( + variant: LanguageVariantModels.ContentItemLanguageVariant, + workflows: ReadonlyArray, +): string => { + const found = findVariantStep(variant, workflows); + if (found === undefined) { + return `unknown workflow ${variant.workflow.workflowIdentifier.id ?? "?"} step ${variant.workflow.stepIdentifier.id ?? "?"}`; + } + + return `${found.workflow.codename}/${found.step.codename}`; +}; + +type WorkflowStep = Readonly<{ id: string; name: string; codename: string }>; + +const findVariantStep = ( + variant: LanguageVariantModels.ContentItemLanguageVariant, + workflows: ReadonlyArray, +): Readonly<{ workflow: WorkflowModels.Workflow; step: WorkflowStep }> | undefined => { + const workflow = workflows.find( + (candidate) => candidate.id === variant.workflow.workflowIdentifier.id, + ); + if (workflow === undefined) { + return undefined; + } + + const step = [ + ...workflow.steps, + workflow.publishedStep, + workflow.scheduledStep, + workflow.archivedStep, + ].find((candidate) => candidate.id === variant.workflow.stepIdentifier.id); + if (step === undefined) { + return undefined; + } + + return { workflow, step }; +}; + +const hasStringMessage = (cause: unknown): cause is { message: string } => + typeof cause === "object" && + cause !== null && + "message" in cause && + typeof (cause as { message: unknown }).message === "string"; diff --git a/evals/lib/invocations.ts b/evals/lib/invocations.ts new file mode 100644 index 0000000..0e811c6 --- /dev/null +++ b/evals/lib/invocations.ts @@ -0,0 +1,43 @@ +import * as z from "zod/mini"; + +// One line of the shim's log (evals/shim.ts): the CLI process's exit code and argv, key already redacted. +const CliInvocationSchema = z.object({ exitCode: z.int(), args: z.array(z.string()) }); +export type CliInvocation = z.infer; + +export type InvocationCounts = Readonly<{ + cliInvocationCount: number; + failedCliInvocationCount: number; + helpLookupCount: number; + docsLookupCount: number; +}>; + +export const parseInvocationLog = (log: string): ReadonlyArray => + log.split("\n").flatMap(parseLine); + +export const countInvocations = (invocations: ReadonlyArray): InvocationCounts => ({ + cliInvocationCount: invocations.length, + failedCliInvocationCount: invocations.filter((invocation) => invocation.exitCode !== 0).length, + helpLookupCount: invocations.filter(isHelpLookup).length, + docsLookupCount: invocations.filter(isDocsLookup).length, +}); + +const parseLine = (line: string): ReadonlyArray => { + if (line === "") { + return []; + } + const parsed = z.safeParse(CliInvocationSchema, tryParseJson(line)); + return parsed.success ? [parsed.data] : []; +}; + +const tryParseJson = (line: string): unknown => { + try { + return JSON.parse(line); + } catch { + return undefined; + } +}; + +const isHelpLookup = (invocation: CliInvocation): boolean => + invocation.args.includes("--help") || invocation.args.includes("-h"); + +const isDocsLookup = (invocation: CliInvocation): boolean => invocation.args[0] === "docs"; diff --git a/evals/lib/messages.ts b/evals/lib/messages.ts new file mode 100644 index 0000000..05211d2 --- /dev/null +++ b/evals/lib/messages.ts @@ -0,0 +1,6 @@ +import type { SDKMessage, SDKResultMessage } from "@anthropic-ai/claude-agent-sdk"; + +export const findResultMessage = ( + messages: ReadonlyArray, +): SDKResultMessage | undefined => + messages.findLast((message): message is SDKResultMessage => message.type === "result"); diff --git a/evals/lib/policy.ts b/evals/lib/policy.ts new file mode 100644 index 0000000..a405019 --- /dev/null +++ b/evals/lib/policy.ts @@ -0,0 +1,145 @@ +import type { BashInput, WebFetchInput } from "@anthropic-ai/claude-agent-sdk/sdk-tools"; +import * as z from "zod/mini"; +import { isNone, isSome, none, type Option, some } from "../../src/lib/option.js"; +import { err, isErr, ok, type Result } from "../../src/lib/result.js"; + +export const AGENT_TOOLS: ReadonlyArray = ["Bash", "WebFetch"]; + +export const DENIAL_PREFIX = "blocked by policy: "; + +// `satisfies` fails typecheck when the SDK renames or retypes a field the schema reads. +export const bashInputSchema = z.object({ command: z.string() }) satisfies z.ZodMiniType< + Pick +>; +export const webFetchInputSchema = z.object({ + url: z.string(), + prompt: z.string(), +}) satisfies z.ZodMiniType; + +const WEB_FETCH_ALLOWED_HOSTS: ReadonlyArray = ["kontent.ai"]; + +export const AGENT_PERMISSION_RULES: ReadonlyArray = [ + "Bash", + ...WEB_FETCH_ALLOWED_HOSTS.map((host) => `WebFetch(domain:${host})`), +]; + +export type ToolPolicy = Readonly<{ + workspaceDir: string; + mapiKey: string; +}>; + +export type ToolCallRequest = Readonly<{ toolName: string; input: unknown }>; + +export const applyToolPolicy = ( + policy: ToolPolicy, + request: ToolCallRequest, +): Result => { + if (request.toolName === "Bash") { + const parsed = bashInputSchema.safeParse(request.input); + if (!parsed.success) { + return err(`${DENIAL_PREFIX}malformed Bash input`); + } + const deniedToken = findPathOutsideWorkspace( + parsed.data.command, + workspaceDirAliases(policy.workspaceDir), + ); + return isSome(deniedToken) + ? err(`${DENIAL_PREFIX}command references a path outside the workspace: ${deniedToken.value}`) + : ok(undefined); + } + if (request.toolName === "WebFetch") { + const parsed = webFetchInputSchema.safeParse(request.input); + if (!parsed.success) { + return err(`${DENIAL_PREFIX}malformed WebFetch input`); + } + const violation = checkWebFetch(policy, parsed.data.url); + return isErr(violation) ? err(`${DENIAL_PREFIX}${violation.error}`) : ok(undefined); + } + return ok(undefined); +}; + +// Conservative and simple by design: flags a token as escaping the workspace +// directory if it is home-relative, walks up via `..`, or is an absolute path +// under a filesystem root that holds user or system data. Bare API paths like +// `/types` and `/dev/null` stay allowed. False positives (denying something +// safe) are acceptable; false negatives are not. +export const findPathOutsideWorkspace = ( + command: string, + workspaceDirs: ReadonlyArray, +): Option => { + const token = command + .split(/[\s<>|;&()]+/) + .filter((token) => token !== "") + .find((token) => isPathOutsideWorkspace(token, workspaceDirs)); + return token === undefined ? none : some(token); +}; + +// macOS hands out `/var/folders/...` from mkdtemp while `pwd` inside it +// reports `/private/var/folders/...`; both spellings count as inside. +export const workspaceDirAliases = (workspaceDir: string): ReadonlyArray => + workspaceDir.startsWith("/private/") + ? [workspaceDir, workspaceDir.slice("/private".length)] + : [workspaceDir, `/private${workspaceDir}`]; + +// The host check backs the WebFetch(domain:...) rule, whose subdomain and redirect handling is undocumented. +export const checkWebFetch = (policy: ToolPolicy, url: string): Result => { + if (policy.mapiKey !== "" && url.includes(policy.mapiKey)) { + return err("url contains the Management API key"); + } + const host = parseHost(url); + if (isNone(host)) { + return err("url is not a valid http(s) url"); + } + if (!WEB_FETCH_ALLOWED_HOSTS.includes(host.value)) { + return err(`host is not allowed: ${host.value}`); + } + return ok(undefined); +}; + +const FILESYSTEM_ROOTS: ReadonlyArray = [ + "/Users/", + "/home/", + "/root/", + "/etc/", + "/var/", + "/private/", + "/tmp/", + "/opt/", + "/usr/", + "/Library/", + "/Applications/", + "/Volumes/", + "/proc/", + "/sys/", + "/mnt/", +]; + +const isPathOutsideWorkspace = (token: string, workspaceDirs: ReadonlyArray): boolean => { + const path = token.replace(/^["']|["']$/g, ""); + if ( + path.startsWith("~") || + path.includes("$HOME") || + path.includes(`\${HOME}`) || + path.split("/").includes("..") + ) { + return true; + } + // A root's trailing slash means `path.startsWith(root)` misses the bare + // root itself (e.g. `/etc` from `ls /etc`), so also match it without one. + const isUnderRoot = FILESYSTEM_ROOTS.some( + (root) => path === root.slice(0, -1) || path.startsWith(root), + ); + const isInsideWorkspace = workspaceDirs.some((dir) => path === dir || path.startsWith(`${dir}/`)); + return isUnderRoot && !isInsideWorkspace; +}; + +const parseHost = (url: string): Option => { + try { + const parsed = new URL(url); + return parsed.protocol === "http:" || parsed.protocol === "https:" + ? some(parsed.hostname.toLowerCase()) + : none; + } catch { + return none; + } +}; diff --git a/evals/lib/prompt.ts b/evals/lib/prompt.ts new file mode 100644 index 0000000..9eceefd --- /dev/null +++ b/evals/lib/prompt.ts @@ -0,0 +1,21 @@ +export type BuildTaskPromptParams = Readonly<{ + envId: string; + workspaceDir: string; + taskPrompt: string; +}>; + +export const buildPreamble = (envId: string, workspaceDir: string): string => + [ + `You are working with a Kontent.ai environment. Its id is ${envId}.`, + "A Management API key for it is in the environment variable EVALS_MAPI_KEY. Never print, echo or paste it anywhere.", + "The `kontent` CLI is installed and on PATH. Use it for everything you do with Kontent.ai.", + `Work only inside ${workspaceDir}. You are forbidden to read, search, list or open any file or directory outside it, including source code, configuration and home directories on this machine. This includes commands like cat, ls, find, grep on other paths.`, + `Any file you create - request bodies, scratch files, notes - must also live inside ${workspaceDir}. Do not use /tmp or your home directory.`, + "When finished, reply with: every command you ran in order, what was unclear or annoying, and anything you had to guess.", + ].join("\n"); + +export const buildTaskPrompt = ({ + envId, + workspaceDir, + taskPrompt, +}: BuildTaskPromptParams): string => `${buildPreamble(envId, workspaceDir)}\n\n${taskPrompt}`; diff --git a/evals/lib/registry.ts b/evals/lib/registry.ts new file mode 100644 index 0000000..dad9d82 --- /dev/null +++ b/evals/lib/registry.ts @@ -0,0 +1,18 @@ +import { contentTypeWithSnippet } from "../tasks/content-type-with-snippet/task.js"; +import { publishItem } from "../tasks/publish-item/task.js"; +import { schedulePublish } from "../tasks/schedule-publish/task.js"; +import { taxonomyGroup } from "../tasks/taxonomy-group/task.js"; +import { updateContentType } from "../tasks/update-content-type/task.js"; +import { updateLanguageVariant } from "../tasks/update-language-variant/task.js"; +import { workflowAndCollection } from "../tasks/workflow-and-collection/task.js"; +import type { EvalTask } from "./types.js"; + +export const evalTasks: ReadonlyArray = [ + taxonomyGroup, + workflowAndCollection, + contentTypeWithSnippet, + publishItem, + schedulePublish, + updateLanguageVariant, + updateContentType, +]; diff --git a/evals/lib/report/markdown.ts b/evals/lib/report/markdown.ts new file mode 100644 index 0000000..f6286d6 --- /dev/null +++ b/evals/lib/report/markdown.ts @@ -0,0 +1,206 @@ +import type { RunSummary, RunSummaryHeader, RunSummaryRow } from "../results.js"; +import type { ToolCall, ToolCallOutcome } from "../toolCalls.js"; +import type { TaskTrace } from "../transcript.js"; +import type { Assertion, Verdict } from "../types.js"; + +// Nice-to-have, not load-bearing: every function here is a pure string +// renderer with no LLM calls and no network access, so deleting this whole +// folder (and the one call into it from results.ts) removes the feature +// cleanly. + +export const renderTaskReport = ( + taskId: string, + trace: TaskTrace, + verdict: Verdict, + assertions: ReadonlyArray, + errorMessage?: string, +): string => + [ + `# ${taskId}: ${verdict}`, + "", + renderStatsLine(trace), + "", + ...renderErrorSection(errorMessage), + "## Assertions", + ...renderAssertions(assertions), + "", + "## Tool calls", + ...renderToolCalls(trace), + "", + "## Denied", + ...renderDenied(trace.toolCalls), + "", + "## Agent final reply", + trace.finalReply, + "", + ].join("\n"); + +export const renderRunReport = ( + summary: RunSummary, + traces: ReadonlyArray>, +): string => + [ + renderHeader(summary.header), + "", + renderTable(summary.rows), + "", + renderTotals(summary.rows), + "", + "## Friction", + "", + "### Failed calls", + ...renderFailedCalls(traces), + "", + "### Web fetches", + ...renderWebFetches(traces), + "", + "## Agent final replies", + ...renderFinalReplies(traces), + "", + ].join("\n"); + +const renderStatsLine = (trace: TaskTrace): string => + `turns ${trace.numbers.turns} | calls ${trace.toolCalls.length} | cli ${trace.cliInvocationCount} | failed ${trace.failedCliInvocationCount} | denied ${trace.deniedCallCount} | ` + + `help ${trace.helpLookupCount} | docs ${trace.docsLookupCount} | fetch ${trace.webFetchCount} | ` + + `cost $${trace.numbers.costUsd.toFixed(2)} | ` + + `tokens in ${trace.numbers.inputTokens} | out ${trace.numbers.outputTokens} | cache ${trace.numbers.cacheReadTokens + trace.numbers.cacheCreationTokens} | ` + + `${formatDuration(trace.numbers.durationMs)} | stop: ${trace.stopReason}`; + +const renderErrorSection = (errorMessage: string | undefined): ReadonlyArray => + errorMessage === undefined ? [] : ["## Error", "", errorMessage, ""]; + +const renderAssertions = (assertions: ReadonlyArray): ReadonlyArray => + assertions.length === 0 + ? ["(none)"] + : assertions.map( + (assertion) => + `- ${assertion.passed ? "PASS" : "FAIL"} ${assertion.id} ${assertion.evidence}`, + ); + +const renderToolCalls = (trace: TaskTrace): ReadonlyArray => { + const beforeFirstCall = renderAgentTextsAfter(trace, -1); + const calls = + trace.toolCalls.length === 0 + ? ["(none)"] + : trace.toolCalls.flatMap((call) => renderToolCall(call, trace)); + return [...beforeFirstCall, ...calls]; +}; + +const renderAgentTextsAfter = ( + trace: TaskTrace, + afterToolCallIndex: number, +): ReadonlyArray => + trace.agentTexts + .filter((text) => text.afterToolCallIndex === afterToolCallIndex) + .map((text) => `> agent: "${text.text}"`); + +const STATUS_LABELS: Readonly> = { + ok: "ok", + failed: "FAIL", + denied: "DENIED", + "no-result": "NORESULT", +}; + +const renderToolCall = (call: ToolCall, trace: TaskTrace): ReadonlyArray => { + const status = STATUS_LABELS[call.outcome].padEnd(8); + const duration = formatDuration(call.durationMs).padStart(6); + const header = `${call.index + 1}. ${status} ${duration} ${renderInvocation(call)}`; + const errorLine = call.outcome === "ok" ? [] : [` ${firstErrorLine(call)}`]; + const replies = trace.agentTexts + .filter((text) => text.afterToolCallIndex === call.index) + .map((text) => ` > agent: "${text.text}"`); + + return [header, ...errorLine, ...replies]; +}; + +const renderInvocation = (call: ToolCall): string => + call.tool === "WebFetch" + ? `WebFetch ${call.command} ("${call.fetchPrompt}")` + : `\`${call.command}\``; + +// Bash commands can be multi-line heredocs; only the first line is shown so a +// failed call's body (e.g. JSON piped into a file) doesn't flood the report. +const renderFirstLine = (call: ToolCall): string => + call.tool === "WebFetch" ? renderInvocation(call) : `\`${call.command.split("\n")[0]?.trim()}\``; + +const firstErrorLine = (call: ToolCall): string => + firstNonEmptyLine(call.stderr) ?? firstNonEmptyLine(call.stdout) ?? "(no output)"; + +const firstNonEmptyLine = (text: string): string | undefined => + text + .split("\n") + .map((line) => line.trim()) + .find((line) => line !== ""); + +const renderDenied = (toolCalls: ReadonlyArray): ReadonlyArray => { + const denied = toolCalls.filter((call) => call.outcome === "denied"); + return denied.length === 0 + ? ["(none)"] + : denied.map((call) => `- ${call.tool} \`${call.command}\` (${call.stderr})`); +}; + +const renderHeader = (header: RunSummaryHeader): string => + `# Eval run: ${header.model} @ ${header.envId} (${header.startedAt})\n\n` + + `cli ${header.cliVersion} | git ${header.gitSha} | tools ${header.tools.join(", ")} | ` + + `rules ${header.permissionRules.join(", ")} | max turns ${header.maxTurns} | ` + + `preamble ${header.preambleHash}`; + +const renderTable = (rows: ReadonlyArray): string => + [ + "| task | verdict | turns | calls | cli | failed | denied | help | docs | fetch | cost | in | out | time |", + "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |", + ...rows.map( + (row) => + `| ${row.id} | ${row.verdict} | ${row.turns} | ${row.toolCallCount} | ${row.cliInvocationCount} | ${row.failedCliInvocationCount} | ` + + `${row.deniedCallCount} | ${row.helpLookupCount} | ${row.docsLookupCount} | ${row.webFetchCount} | ` + + `$${row.costUsd.toFixed(2)} | ${formatThousands(row.inputTokens)} | ${formatThousands(row.outputTokens)} | ${formatDuration(row.durationMs)} |`, + ), + ].join("\n"); + +const formatThousands = (count: number): string => `${Math.round(count / 1000)}k`; + +const renderTotals = (rows: ReadonlyArray): string => { + const passCount = rows.filter((row) => row.verdict === "PASS").length; + const totalCost = rows.reduce((sum, row) => sum + row.costUsd, 0); + const totalDuration = rows.reduce((sum, row) => sum + row.durationMs, 0); + return `${passCount}/${rows.length} passed | total cost $${totalCost.toFixed(2)} | total time ${formatDuration(totalDuration)}`; +}; + +// Every failure listed, not grouped: with single-digit counts per run, a prefix heuristic hides more than it shows. +const renderFailedCalls = ( + traces: ReadonlyArray>, +): ReadonlyArray => { + const failed = traces.flatMap(({ id, trace }) => + trace.toolCalls + .filter((call) => call.outcome === "failed") + .map((call) => `- ${id}: ${renderFirstLine(call)} ${firstErrorLine(call)}`), + ); + return failed.length === 0 ? ["(none)"] : failed; +}; + +// Every fetch listed, not top groups: each one marks where the CLI's own docs did not carry the agent. +const renderWebFetches = ( + traces: ReadonlyArray>, +): ReadonlyArray => { + const fetches = traces.flatMap(({ id, trace }) => + trace.toolCalls + .filter((call) => call.tool === "WebFetch") + .map((call) => `- ${id}: ${STATUS_LABELS[call.outcome]} ${renderInvocation(call)}`), + ); + return fetches.length === 0 ? ["(none)"] : fetches; +}; + +const renderFinalReplies = ( + traces: ReadonlyArray>, +): ReadonlyArray => + traces.flatMap(({ id, trace }) => [`### ${id}`, "", trace.finalReply, ""]); + +const formatDuration = (ms: number | null): string => { + if (ms === null) { + return "-"; + } + const totalSeconds = ms / 1000; + return totalSeconds >= 60 + ? `${Math.floor(totalSeconds / 60)}m ${Math.round(totalSeconds % 60)}s` + : `${totalSeconds.toFixed(1)}s`; +}; diff --git a/evals/lib/results.ts b/evals/lib/results.ts new file mode 100644 index 0000000..ad5691d --- /dev/null +++ b/evals/lib/results.ts @@ -0,0 +1,177 @@ +import { execFile } from "node:child_process"; +import { createHash } from "node:crypto"; +import { mkdir, readFile, writeFile } from "node:fs/promises"; +import { promisify } from "node:util"; +import type { AgentRun } from "./agent.js"; +import { AGENT_PERMISSION_RULES, AGENT_TOOLS } from "./policy.js"; +import { buildPreamble } from "./prompt.js"; +import { renderRunReport, renderTaskReport } from "./report/markdown.js"; +import type { TaskTrace } from "./transcript.js"; +import type { Assertion, Verdict } from "./types.js"; + +const execFileAsync = promisify(execFile); + +export type RunSummaryHeader = Readonly<{ + model: string; + tools: ReadonlyArray; + permissionRules: ReadonlyArray; + maxTurns: number; + cliVersion: string; + gitSha: string; + envId: string; + preambleHash: string; + startedAt: string; + finishedAt: string; +}>; + +export type RunSummaryRow = Readonly<{ + id: string; + verdict: Verdict; + turns: number; + toolCallCount: number; + cliInvocationCount: number; + failedCliInvocationCount: number; + deniedCallCount: number; + helpLookupCount: number; + docsLookupCount: number; + webFetchCount: number; + costUsd: number; + inputTokens: number; + outputTokens: number; + durationMs: number; + stopReason: string; +}>; + +export type RunSummary = Readonly<{ + header: RunSummaryHeader; + rows: ReadonlyArray; +}>; + +export type TaskResultInput = Readonly<{ + trace: TaskTrace; + run: AgentRun | undefined; + verdict: Verdict; + assertions: ReadonlyArray; + // Set for ERROR verdicts: the agent-run or check failure that produced them, + // so it survives on disk instead of only ever appearing in a thrown Error. + errorMessage?: string; +}>; + +// Creates evals/results/-