diff --git a/.licenserc.yaml b/.licenserc.yaml index 5c048ea152..7fc9cd170a 100644 --- a/.licenserc.yaml +++ b/.licenserc.yaml @@ -43,6 +43,7 @@ header: - '**/.gitignore' - '**/.dockerignore' - '**/.env.example' + - '**/*.example.env' - '**/LICENSE' - 'NOTICE' - '**/LICENSE.oceanbase-design' diff --git a/evaluation/README.md b/evaluation/README.md index 49d67c955b..e38156e628 100644 --- a/evaluation/README.md +++ b/evaluation/README.md @@ -181,11 +181,14 @@ uv run --project evaluation powercontext-eval swebench-pro create-batch \ --idempotency-key "stability-$(date -u +%Y%m%dT%H%M%SZ)" ``` -## LongMemEval-V2 smoke input validation +## LongMemEval-V2 smoke workload -The LongMemEval-V2 command currently validates a fixed smoke subset and writes -its preflight artifacts. It does not yet invoke a PowerContext Memory adapter, -a Reader, or a Judge. +The LongMemEval-V2 commands run a fixed ten-question smoke subset end to end: preflight +validation of the pinned inputs, real PowerContext HTTP retrieval, prompt preparation, an +optional Reader, upstream scoring, offline score replay, and a one-command orchestration +(`run-smoke`) that writes one run directory with a unified report. A full-tier run has not been +executed; its recorded configuration lives in +[docs/longmemeval-v2-full-run.md](docs/longmemeval-v2-full-run.md). Prepare a detached upstream checkout at the pinned harness commit and download the matching LongMemEval-V2 data root outside this repository. The checked-in @@ -227,6 +230,309 @@ The command refuses a harness checkout at a different commit, mismatched input hashes, invalid or incomplete smoke coverage, and an existing output directory. It writes `manifest.json` and `subset.json`, both labelled as a smoke subset. +### PowerContext Memory adapter bootstrap + +Run the bootstrap with a Python environment containing the pinned upstream +harness dependencies. It validates the clean harness commit, registers +`memory_type: powercontext` in memory, and then forwards all remaining arguments +to the unchanged upstream harness: + +```bash +python evaluation/scripts/run_longmemeval_v2_harness.py \ + --harness-root /path/to/LongMemEval-V2 \ + -- +``` + +The adapter memory configuration requires a dedicated evaluation Scope and +audit artifact path. Credentials are resolved only from `token_env` at runtime: + +```json +{ + "memory_type": "powercontext", + "memory_params": { + "scope_id": "longmemeval-v2-smoke-run-id", + "audit_path": "/path/to/run/context/powercontext-memory.jsonl", + "base_url": "http://127.0.0.1:8765", + "token_env": "POWERCONTEXT_TOKEN", + "search_mode": "auto", + "search_limit": 10 + } +} +``` + +Each trajectory chunk is captured through the public Content Source endpoint +and one deterministic bounded projection per trajectory is explicitly +remembered through the public Memory endpoint. The projection keeps goal, +outcome, URL, action, thought, and bounded observation evidence without calling +a model. The adapter audit correlates the returned Source references and Memory +citation; it does not claim that this correlation is native Memory lineage. +Queries use the public Memory search endpoint and return upstream-compatible +text context items. Query images are neither read nor sent because the current +Memory search contract is text only. + +With a ready PowerContext Server, run the fixed ten-question retrieval-only +smoke workload without a Reader, Judge, model credential, or scoring step: + +```bash +uv run --project evaluation powercontext-eval longmemeval-v2 retrieval-smoke \ + --harness-root /path/to/LongMemEval-V2 \ + --data-root /path/to/longmemeval-v2-data \ + --dataset-lock evaluation/locks/longmemeval-v2-small-v1.dataset-lock.json \ + --smoke-manifest evaluation/locks/longmemeval-v2-small-v1.smoke.json \ + --base-url http://127.0.0.1:8000 \ + --run-id retrieval-smoke-001 \ + --powercontext-revision POWERCONTEXT_GIT_SHA \ + --integration-revision INTEGRATION_GIT_SHA \ + --output-dir /path/to/new-retrieval-smoke-artifacts +``` + +The runner verifies the locked input digests while streaming +`trajectories.jsonl`, retains only one trajectory object at a time, and creates +one isolated child Scope for each distinct ordered haystack. Identical +haystacks reuse their ingestion, while different haystacks cannot retrieve each +other's Memory. It writes `retrieval-manifest.json`, +`retrieval-results.jsonl`, `adapter-audit.jsonl`, `failures.jsonl`, and +`summary.json` beside the preflight `manifest.json` and `subset.json`. Accuracy, +Reader, and Judge fields remain null because retrieval-only output is not a +benchmark score. + +### Experiment arms + +Retrieval behaviour is selected through a registered experiment arm, not a free-form search mode: + +```bash +uv run --project evaluation powercontext-eval longmemeval-v2 retrieval-smoke \ + ... \ + --experiment-arm current-memory-hybrid-v1 +``` + +An arm is a frozen configuration identity: same questions and the same downstream token +budget, with only the declared retrieval/projection knobs allowed to differ. Five arms are +currently registered: + +| Arm ID | Strategy | Notes | +| --- | --- | --- | +| `current-memory-fts-v1` | Memory search (`fts`) | Default; identical to the previous `--search-mode fts` behaviour. | +| `current-memory-hybrid-v1` | Memory search (`hybrid`) | Requires a Server whose `/v1/capabilities` advertises `hybrid`. | +| `query-time-compact-v1` | PreparedContext (8,000 bytes) | Uses public `/v1/context/prepare`; does not alter ingestion or persist a new index schema. | +| `write-time-l0-l1-v1` | Memory search (`fts`) | Writes deterministic bounded L0 index and L1 summary entries through public `remember`; no schema fields are added. | +| `task-lensed-selection-v1` | Memory search (`fts`) | Projects the question into a deterministic keyword lens; never reads question type, gold answer, or Judge data. | + +Unregistered arm IDs (currently the unsupported temporal-filter arm) are rejected +before any work runs instead of silently falling back to FTS. Before ingesting anything, the +runner checks `/v1/capabilities`: a Server without the arm's search mode or PreparedContext +schema fails as a capability error. Every explicitly requested search mode must match the +executed mode the Server reports (only `auto` accepts the Server's own choice) — a mismatch is +recorded as an integrity failure rather than a benchmark result. Every manifest, summary, and +unified report records the full arm block, `adapter-audit.jsonl` records the requested and +actual strategy/mode per query, and `ensure_comparable_experiment_runs` refuses to compare two +runs whose dataset lock, question manifest, harness commit, processor, context budget, search +limit, Reader/Judge configuration, or revisions differ beyond the arm. + +Prepare bounded, replayable Reader inputs without calling a Reader. This command +uses the pinned upstream harness to count and truncate Memory context, and +requires an immutable Hugging Face processor revision rather than resolving the +processor from a moving branch: + +```bash +uv run --project evaluation powercontext-eval longmemeval-v2 prepare-smoke \ + --retrieval-dir /path/to/retrieval-smoke-artifacts \ + --harness-root /path/to/LongMemEval-V2 \ + --harness-python /path/to/longmemeval-harness-python \ + --processor-revision PROCESSOR_GIT_SHA \ + --memory-context-max-tokens 200000 \ + --output-dir /path/to/new-prepared-prompt-artifacts +``` + +It writes `prepare-manifest.json`, `prepared-prompts.jsonl`, +`prepare-failures.jsonl`, and `prepare-summary.json`. Each prepared prompt +records the original and bounded Context token counts, the bounded Context +bytes, final system/user messages, a prompt SHA-256, and prepare latency. This +is still a no-model stage: Reader, Judge, and accuracy remain absent. + +Run an Anthropic-compatible Reader or DeepSeek's direct OpenAI-compatible API +over prepared prompts. Credentials are read only from the named process +environment variable and are never written to the run artifacts: + +```bash +export DEEPSEEK_API_KEY=YOUR_KEY +uv run --project evaluation powercontext-eval longmemeval-v2 reader-smoke \ + --prepared-dir /path/to/prepared-prompt-artifacts \ + --provider deepseek-openai \ + --model deepseek-flash \ + --token-env DEEPSEEK_API_KEY \ + --max-tokens 512 \ + --temperature 0 \ + --output-dir /path/to/new-reader-artifacts +``` + +The Reader writes `reader-manifest.json`, `reader-outputs.jsonl`, +`reader-failures.jsonl`, and `reader-summary.json`. It records provider usage +returned by the API but does not score answers. Keep OpenTelemetry variables +out of this command when prompt telemetry is not approved. + +Score Reader outputs with the pinned deterministic metric functions. The +abstention and gotchas metric types require an LLM Judge and send the question, +reference answer, full Reader response, and parsed answer to the configured +Judge provider: + +```bash +export DEEPSEEK_API_KEY=YOUR_KEY +uv run --project evaluation powercontext-eval longmemeval-v2 score-smoke \ + --reader-dir /path/to/reader-artifacts \ + --data-root /path/to/longmemeval-v2-data \ + --dataset-lock evaluation/locks/longmemeval-v2-small-v1.dataset-lock.json \ + --smoke-manifest evaluation/locks/longmemeval-v2-small-v1.smoke.json \ + --harness-root /path/to/LongMemEval-V2 \ + --judge-model deepseek-flash \ + --judge-token-env DEEPSEEK_API_KEY \ + --judge-max-tokens 256 \ + --judge-temperature 0 \ + --output-dir /path/to/new-score-artifacts +``` + +The score run writes `per-question.jsonl`, `judge-outputs.jsonl`, +`score-failures.jsonl`, and `score-summary.json`. It also writes +`scoring-inputs.local.jsonl`, which contains reference answers for local replay +and must remain outside Git, shared reports, and telemetry. + +Replay the saved deterministic and Judge decisions without a Reader, Judge +provider, token, or network request: + +```bash +uv run --project evaluation powercontext-eval longmemeval-v2 replay-score \ + --score-dir /path/to/score-artifacts \ + --harness-root /path/to/LongMemEval-V2 \ + --output-dir /path/to/new-score-replay-artifacts +``` + +The replay records source digests and writes `replay-manifest.json`, +`replay-per-question.jsonl`, `replay-failures.jsonl`, and +`replay-summary.json`. It reuses only saved Judge labels for LLM-scored cases. + +### One-command smoke run + +`run-smoke` chains every stage above into one fail-closed run directory. With a +ready PowerContext Server and no model credentials, run the model-free mode +first. The retrieval arm defaults to `current-memory-fts-v1` and can be selected +explicitly with `--experiment-arm`: + +```bash +uv run --project evaluation powercontext-eval longmemeval-v2 run-smoke \ + --data-root /path/to/longmemeval-v2-data \ + --dataset-lock evaluation/locks/longmemeval-v2-small-v1.dataset-lock.json \ + --smoke-manifest evaluation/locks/longmemeval-v2-small-v1.smoke.json \ + --harness-root /path/to/LongMemEval-V2 \ + --harness-python /path/to/longmemeval-harness-python \ + --processor-revision PROCESSOR_GIT_SHA \ + --powercontext-revision POWERCONTEXT_GIT_SHA \ + --integration-revision INTEGRATION_GIT_SHA \ + --powercontext-base-url http://127.0.0.1:18765 \ + --experiment-arm current-memory-fts-v1 \ + --skip-reader \ + --output-dir /path/to/new-run-artifacts +``` + +The full mode drops `--skip-reader`, requires `DEEPSEEK_API_KEY` in the environment, and also +runs the Reader, Judge scoring, and score replay. `--skip-score` keeps the Reader but skips +Judge scoring and replay. Every mode writes the same layout: + +```text +/ + run-manifest.json # exclusive-create; inputs, revisions, providers (no secret values) + run-summary.json # completed/skipped/failed phases; accuracy only when scored + failures.jsonl # one row per failed phase, with an error class + report.json # unified machine-readable summary + report.md # human summary with the fixed boundary banner + 01-inputs/ 02-retrieval/ 03-prepare/ 04-reader/ 05-score/ 06-replay/ +``` + +The command exits non-zero unless the run completed: `--skip-reader` and `--skip-score` runs end +as `partial`, and a failed stage ends as `failed` while keeping the earlier stage artifacts. +Recorded failure summaries are redacted against the configured token values, including +`POWERCONTEXT_TOKEN` in model-free mode. Reference answers are read only by the score stage, +which writes the local replay artifact, and by the replay stage reading that artifact; the +adapter, retrieval, prepare, and reader stages never read them. + +### Cost reporting with an explicit price policy + +Token usage is always recorded from the provider's own response. Model cost is reported only when +an explicit price policy is passed; prices are never hardcoded, and an unconfigured cost stays +`null` with a reason instead of `0`: + +```bash +uv run --project evaluation powercontext-eval longmemeval-v2 run-smoke \ + ... \ + --price-policy '{"provider":"deepseek-openai","model":"deepseek-flash","currency":"USD","input_cache_hit_price_per_million":0.006,"input_cache_miss_price_per_million":0.3,"output_price_per_million":1.2,"price_policy_revision":"deepseek-public-list-2026-09"}' \ + --judge-price-policy '{"provider":"deepseek-openai","model":"deepseek-flash","currency":"USD","input_cache_hit_price_per_million":0.006,"input_cache_miss_price_per_million":0.3,"output_price_per_million":1.2,"price_policy_revision":"deepseek-public-list-2026-09"}' \ + --output-dir /path/to/new-run-artifacts +``` + +`run-smoke` takes a separate policy per model role because the Reader and the Judge may run +different models; `reader-smoke` takes `--price-policy` and `score-smoke` takes +`--judge-price-policy`. A policy must contain exactly `provider`, `model`, `currency`, +`input_cache_hit_price_per_million`, `input_cache_miss_price_per_million`, +`output_price_per_million`, and `price_policy_revision`. Every field is required, prices must be +finite and non-negative, and `currency` must be `USD` because the recorded amount fields are named +`*_usd`. A policy is applied only when both its `provider` and its `model` match the configured +stage; otherwise the stage records `null` plus the mismatch reason instead of borrowing an +unrelated price. + +Cached and uncached input are priced separately. DeepSeek reports +`prompt_cache_hit_tokens` and `prompt_cache_miss_tokens` alongside `prompt_tokens`, and the +cache-hit rate is roughly fifty times cheaper than the cache-miss rate, so the reader and judge +records keep that split instead of collapsing it into one blended input total. A response that +reports no split cannot be priced at either rate, so its stage records `null` with that reason +rather than an estimate. + +Reader cost comes from the Reader's reported usage, Judge cost from the Judge's reported usage, +and each stage summary and manifest records the policy identity it was priced under. The unified +report adds `usage.reader_cost`, `usage.judge_cost`, `usage.estimated_cost_usd` (the sum, only +when every model stage that actually ran was priced in USD under one policy revision) and +`usage.estimated_cost_note` explaining how the total was computed or why it stays `null`. The +report decides which stages ran from the run manifest `modes` plus the stage summaries, so a +missing, unpriced, mismatched, or differently-revisioned stage cost keeps the total `null` +instead of silently summing the stages that happen to be present. Ingestion is model-free, so the +report records `usage.ingestion_cost` as zero tokens, `cost_usd: 0.0`, and that reason. Without a +policy the report keeps the `null` behavior and explains it. + +Regenerate or inspect a report for any saved run without a model, provider, or server: + +```bash +uv run --project evaluation powercontext-eval longmemeval-v2 report \ + --run-dir /path/to/saved-run \ + --output-dir /path/to/new-report-artifacts +``` + +### LongMemEval-V2 full run (not executed) + +A full-tier run has not been executed and is not approved, and its results must never be +confused with the smoke subset. The recorded prerequisites, configuration template, +cost-estimation and approval gates, output isolation, result labels, and failure-recovery +procedure live in [docs/longmemeval-v2-full-run.md](docs/longmemeval-v2-full-run.md); the +secret-free environment template lives in +[deploy/longmemeval-v2-run-config.example.env](deploy/longmemeval-v2-run-config.example.env). A +full-tier dataset lock and question manifest do not exist yet and must be created and reviewed +before any full run. + +### What the LongMemEval-V2 workload evaluates + +The smoke workload and any future full run measure one pipeline over fixed LongMemEval-V2 +trajectories: whether the PowerContext Memory adapter retrieves citable evidence through public +interfaces under fixed upstream data, fixed questions, and a fixed context budget; the answer +accuracy, latency, context size, failures, and abstention of the configured Reader and Judge; and +whether saved outputs replay the same deterministic scoring inputs without a model. + +They do not evaluate Handoff, cross-host recovery, normal Runtime persistence, or Work +Continuity. A ten-question smoke subset cannot be extrapolated to full benchmark performance, +general model capability, or product leadership, and it does not replace LoCoMo, SWE-bench Pro, +or real user-task acceptance. Scores must never be improved by rewriting gold prompts, reference +answers, or the Memory schema. Every smoke artifact is labelled `smoke-subset`; none of them is +a complete benchmark result. + +The implemented scope and local validation evidence are summarized in +[docs/longmemeval-v2-smoke-delivery.md](docs/longmemeval-v2-smoke-delivery.md). + ## Configuration files Only the environment file configures the evaluation platform. The other files either belong to Codex or are diff --git a/evaluation/deploy/longmemeval-v2-run-config.example.env b/evaluation/deploy/longmemeval-v2-run-config.example.env new file mode 100644 index 0000000000..4494edb2a2 --- /dev/null +++ b/evaluation/deploy/longmemeval-v2-run-config.example.env @@ -0,0 +1,26 @@ +# LongMemEval-V2 runner environment — example, names and placeholders only. +# +# Copy to a protected location outside this repository, fill in the values, and keep the +# file out of Git, logs, reports, and telemetry. Never put a real token in this file, +# on the command line, or in a run artifact. Values below are placeholders. + +# --- PowerContext Memory adapter (used by the retrieval stage) ------------------- +# Bearer token for the local PowerContext Server. Required by every run mode, +# including --skip-reader, because the retrieval stage always talks to the server. +POWERCONTEXT_TOKEN=set-outside-repo + +# --- Reader / Judge, deepseek-openai provider ---------------------------------- +# DeepSeek API key for the Reader (deepseek-openai) and the Judge. +# Required for full runs; not required for --skip-reader runs. +DEEPSEEK_API_KEY=set-outside-repo + +# --- Reader, anthropic-compatible provider (optional) --------------------------- +# Only used when --reader-provider anthropic-compatible is passed. The base URL must +# not contain credentials. +# ANTHROPIC_BASE_URL=https://your-anthropic-compatible-endpoint.example +# ANTHROPIC_AUTH_TOKEN=set-outside-repo + +# --- Optional execution controls ------------------------------------------------ +# Provider timeouts, models, and limits are passed as command-line options +# (--reader-model, --judge-model, --reader-timeout-seconds, ...); see +# evaluation/docs/longmemeval-v2-full-run.md for the recorded configuration template. diff --git a/evaluation/docs/longmemeval-v2-full-run.md b/evaluation/docs/longmemeval-v2-full-run.md new file mode 100644 index 0000000000..1dcc325f92 --- /dev/null +++ b/evaluation/docs/longmemeval-v2-full-run.md @@ -0,0 +1,192 @@ +# LongMemEval-V2 full run configuration (not executed) + +This document records how a **full-tier** LongMemEval-V2 run would be configured and executed. +No full run has been executed. Nothing in this document is a benchmark result, and the pinned +smoke subset described in [../README.md](../README.md) must never be presented as a full result. + +## Status + +| Item | State | +| --- | --- | +| Full-tier dataset lock | **To be created and reviewed.** `evaluation/locks/` currently contains only the small-tier lock `longmemeval-v2-small-v1.dataset-lock.json`. | +| Full-tier question manifest | **To be created and reviewed.** Only the ten-question smoke manifest `longmemeval-v2-small-v1.smoke.json` exists. | +| Cost guards (`--max-cases`, `--max-estimated-cost-usd`) | **Not implemented.** Must be implemented and reviewed before any paid Reader/Judge run (see [Estimates and spend limits](#estimates-and-spend-limits)). | +| Full run execution | **Not executed and not approved.** Model spend and data egress require separate, explicit user approval. | + +## What a full run measures — and what it does not + +A full run measures the same pipeline as the smoke subset, over the complete locked question set: +whether the PowerContext Memory adapter retrieves citable evidence through public interfaces under +fixed upstream data and a fixed context budget, and the answer accuracy, latency, context size, +failures, and abstention of the configured Reader and Judge. + +It does **not** evaluate Handoff, cross-host recovery, normal Runtime persistence, or Work +Continuity; it does not measure general model capability or product leadership, and it does not +replace LoCoMo, SWE-bench Pro, or real user-task acceptance. Scores must never be improved by +rewriting gold prompts, reference answers, or the Memory schema. See +[../README.md](../README.md#what-the-longmemeval-v2-workload-evaluates) for the full boundary list. + +## Prerequisites + +1. **Upstream harness** — a detached checkout of + at commit + `2cc8c540bdb87fe6761629b585e727e1c4704520`, plus a Python environment with that commit's + pinned harness dependencies installed (for example + `D:\powercontext-eval\venvs\longmemeval-v2-2cc8c540`). The runners refuse any other commit. +2. **Dataset** — the LongMemEval-V2 data root downloaded outside this repository (for example + `D:\powercontext-eval\cache\longmemeval-v2-data`). The full run streams + `trajectories.jsonl` (~1.2 GB for the small tier; the full tier is larger), so the disk must + hold the data root plus the run artifacts. A full-tier dataset lock must pin the dataset + revision and the SHA-256 of every input file before the run starts. +3. **PowerContext Server** — a local server bound to loopback (for example + `http://127.0.0.1:18765`) with a bearer token configured. The token is resolved only from the + `POWERCONTEXT_TOKEN` environment variable and is never written to artifacts. +4. **Model providers** — a DeepSeek API key for the `deepseek-openai` Reader and Judge (or an + Anthropic-compatible endpoint for the Reader). No GPU is required; all model access is via + provider APIs. +5. **Network** — access to the local PowerContext Server and to the configured model provider + endpoints. No other egress is required. +6. **Python tooling** — `uv` and the `evaluation` project environment (`uv sync --project evaluation`). + +## Configuration template + +Copy [longmemeval-v2-run-config.example.env](../deploy/longmemeval-v2-run-config.example.env) to a protected +location outside this repository, fill in the values, and load it into the shell before running. +The file contains variable names and placeholders only; never commit a filled copy, and never pass +a token as a command-line argument. + +| Setting | Example | Notes | +| --- | --- | --- | +| Data root | `D:\powercontext-eval\cache\longmemeval-v2-data` | Passed as `--data-root`. | +| Dataset lock | `evaluation\locks\.json` | **Must be created first**; pins dataset revision and file digests. | +| Question manifest | `evaluation\locks\.json` | **Must be created first**; pins the full question set. | +| Harness root | `D:\powercontext-eval\cache\LongMemEval-V2` | Detached checkout at the pinned commit. | +| Harness Python | `D:\powercontext-eval\venvs\longmemeval-v2-2cc8c540\Scripts\python.exe` | Environment with harness dependencies. | +| PowerContext base URL | `http://127.0.0.1:18765` | Loopback only; no credentials in the URL. | +| Reader provider / model | `deepseek-openai` / `deepseek-flash` | Or `anthropic-compatible` with `ANTHROPIC_BASE_URL`. | +| Judge provider / model | `deepseek-openai` / `deepseek-flash` | Only `deepseek-openai` is supported for the Judge. | +| Reader / Judge timeouts | `120.0` seconds each | Raise for slower providers; recorded in the run manifest. | +| Output root | `D:\powercontext-eval\runs\longmemeval-v2\` | Every run creates one new timestamped directory. | + +## Command template + +Once the full-tier lock and manifest exist and are reviewed, a full run uses the same +`run-smoke` command with the full-tier inputs. The paths below are placeholders — **no full-tier +lock exists yet**, so this command must not be run as-is: + +```bash +uv run --project evaluation powercontext-eval longmemeval-v2 run-smoke \ + --data-root D:\powercontext-eval\cache\longmemeval-v2-data \ + --dataset-lock evaluation\locks\.json \ + --smoke-manifest evaluation\locks\.json \ + --harness-root D:\powercontext-eval\cache\LongMemEval-V2 \ + --harness-python D:\powercontext-eval\venvs\longmemeval-v2-2cc8c540\Scripts\python.exe \ + --processor-revision \ + --powercontext-revision \ + --integration-revision \ + --powercontext-base-url http://127.0.0.1:18765 \ + --experiment-arm current-memory-fts-v1 \ + --reader-provider deepseek-openai \ + --reader-model deepseek-flash \ + --judge-provider deepseek-openai \ + --judge-model deepseek-flash \ + --output-dir D:\powercontext-eval\runs\longmemeval-v2\full-v1- +``` + +The retrieval behaviour is selected only through `--experiment-arm` (currently +`current-memory-fts-v1`, `current-memory-hybrid-v1`, `query-time-compact-v1`, or +`write-time-l0-l1-v1`, or `task-lensed-selection-v1`; see +[../README.md](../README.md#experiment-arms)). Two full runs may be compared only when +`ensure_comparable_experiment_runs` confirms that every pinned condition except the arm matches. + +Credentials come only from the environment (`POWERCONTEXT_TOKEN`, `DEEPSEEK_API_KEY`). Without +them the run fails as a `configuration` failure before any model work, with a non-zero exit code. + +## Estimates and spend limits + +A full run spends real money on the Reader and Judge. Before any paid phase: + +1. **Estimate first, without models.** Run `run-smoke` with `--skip-reader` (or the per-stage + `retrieval-smoke` + `prepare-smoke` commands). The prepare stage counts the bounded context + tokens without calling a model; `03-prepare/prepare-summary.json` reports + `question_count` and `memory_context_tokens`. +2. **Compute the estimate.** Reader input tokens ≈ the reported context tokens plus prompt + overhead; Reader output tokens ≤ `--reader-max-tokens` × question count; Judge usage is + reported per question by the provider. Apply the provider's current price table and record + which price table revision was used. +3. **Get explicit approval.** The operator must confirm the estimated cost and the data egress + scope (questions, context, and Reader answers are sent to the configured provider) before the + full Reader/Judge run starts. +4. **Record cost under an explicit price policy.** Pass `--price-policy` (Reader) and + `--judge-price-policy` (Judge) with the operator's current provider prices to have the run price + its real Reader and Judge usage: + + ```powershell + --price-policy '{\"provider\":\"deepseek-openai\",\"model\":\"\",\"currency\":\"USD\",\"input_cache_hit_price_per_million\":,\"input_cache_miss_price_per_million\":,\"output_price_per_million\":,\"price_policy_revision\":\"\"}' + --judge-price-policy '{\"provider\":\"deepseek-openai\",\"model\":\"\",\"currency\":\"USD\",\"input_cache_hit_price_per_million\":,\"input_cache_miss_price_per_million\":,\"output_price_per_million\":,\"price_policy_revision\":\"\"}' + ``` + + Prices are never hardcoded, so the operator supplies and versions them. Each policy must name + both the `provider` and the `model` it prices, and `currency` must be `USD` because the recorded + amount fields are named `*_usd`. DeepSeek bills cached input far below uncached input, so the + policy carries separate cache-hit and cache-miss input prices and the run keeps DeepSeek's + `prompt_cache_hit_tokens`/`prompt_cache_miss_tokens` split; usage that reports no split stays + `null` rather than being blended. Each stage summary and manifest records the policy identity, + and `report.json` reports `usage.reader_cost`, `usage.judge_cost`, and the summed + `usage.estimated_cost_usd` only when every model stage that ran was priced under one policy + revision. +5. **Enforce hard limits.** `--max-cases` and `--max-estimated-cost-usd` are **not implemented** + in this scope, and no generic budget-approval system is provided. Any full run must therefore + be gated by the operator outside the runner: review the step-1 token counts, the step-4 price + policies, and the approval in step 3 before starting the paid Reader/Judge phases. + +`report.json` keeps `estimated_cost_usd` at `null` unless every model stage that actually ran +recorded a usage cost under an explicit price policy. The report decides which stages ran from the +run manifest `modes` and the stage summaries, so a missing, unpriced, provider/model-mismatched, +currency-mismatched, or differently-revisioned stage cost keeps the total `null` with a reason +rather than summing only the stages that are present. Without a policy each stage records +`cost_usd: null` with an `unavailable_reason`, and ingestion — which calls no model — records zero +tokens, `cost_usd: 0.0`, and that reason. + +## Output isolation + +- All artifacts go to a new directory under `D:\powercontext-eval\runs\longmemeval-v2\` (or a + directory the operator explicitly passes). Output directories are fail-closed: an existing + directory aborts the run before any stage starts, and old runs are never overwritten. +- Runs never write to the PowerContext Runtime's normal persistence database; the retrieval + stage creates isolated child Scopes for the evaluation, and query results are recorded in the + run's `adapter-audit.jsonl`. +- Run directories are never committed to Git. In particular, `05-score/scoring-inputs.local.jsonl` + contains reference answers and must stay outside Git, shared reports, and telemetry. + +## Result labels + +| Label | Meaning | +| --- | --- | +| `full` | A full-tier run over the complete locked question set with all six phases completed. Nothing else may be called full. | +| `smoke-subset` | The fixed ten-question subset. Every smoke artifact carries `classification: "smoke-subset"`. | +| `partial` | Phases were skipped (`--skip-reader` / `--skip-score`) or the run stopped after a failure; no accuracy may be published for a partial run. | +| `failed` | A phase failed; `failures.jsonl` records the phase and error class, and the run is never counted as a benchmark score. | + +## Reproducibility + +- `run-manifest.json` records the schema, classification, phase layout, input lock and manifest + SHA-256 digests, harness commit, PowerContext version and URL (without credentials), + Reader/Judge provider and model names (token variable names only, never values), revisions, + and all model parameters. +- `run-summary.json` and `report.json` are derived only from saved stage artifacts; regenerating + a report never calls a model, the provider, or the PowerContext Server. +- `replay-score` re-derives the deterministic scoring inputs and reuses only saved Judge labels + for LLM-scored cases, so the recorded score can be audited without spending tokens. +- To reproduce a run: pin the same harness commit, dataset revision, model IDs, and parameters + recorded in the manifest, then re-run into a **new** output directory. + +## Failure recovery + +- Never overwrite or edit a failed run's directory; it is retained evidence. +- To resume after a failed stage, run the remaining per-stage commands + (`prepare-smoke`, `reader-smoke`, `score-smoke`, `replay-score`) against the completed stage + directories of the interrupted run, writing to new output directories. `run-smoke` itself + always starts from preflight and cannot resume an interrupted run. +- To start over, create a new `run-smoke` output directory; the fail-closed check prevents + accidental reuse. diff --git a/evaluation/docs/longmemeval-v2-smoke-delivery.md b/evaluation/docs/longmemeval-v2-smoke-delivery.md new file mode 100644 index 0000000000..37ffe51701 --- /dev/null +++ b/evaluation/docs/longmemeval-v2-smoke-delivery.md @@ -0,0 +1,141 @@ +# LongMemEval-V2 smoke workload delivery + +This delivery adds a bounded, reproducible LongMemEval-V2 smoke workload for PowerContext. +It uses the public Source and Memory HTTP interfaces, preserves the pinned upstream prompt and +scoring behavior, and labels every result as a smoke subset rather than a complete benchmark. + +## Delivered workflow + +The `powercontext-eval longmemeval-v2` command group now supports the complete local workflow: + +1. validate the pinned dataset, fixed ten-question manifest, and upstream harness checkout; +2. ingest trajectory evidence through public PowerContext Source and Memory APIs; +3. retrieve cited text/image context from isolated evaluation Scopes; +4. prepare bounded Reader messages with the pinned upstream harness implementation; +5. optionally call an explicitly configured Reader and Judge; +6. score with the pinned upstream evaluation functions; +7. replay saved deterministic and Judge decisions without model or network calls; and +8. produce one machine-readable report and one human-readable report. + +`run-smoke` orchestrates these stages into a single fail-closed output directory. An existing +output directory is never overwritten. Infrastructure, retrieval, generation, Judge, +configuration, and integrity failures remain distinct from incorrect answers. + +## Experiment arms + +Retrieval behaviour is selected through a registered experiment arm, not a free-form search +mode. An arm is a frozen experiment identity — same questions, same context budget, same +projections — and only the knobs the arm declares may differ, so two runs of different arms can +be compared fairly. `arms.py` registers exactly the five implemented arms: + +| Arm ID | Strategy | Difference | +| --- | --- | --- | +| `current-memory-fts-v1` | Memory search (`fts`) | Default; identical to the previous `--search-mode fts` behaviour. | +| `current-memory-hybrid-v1` | Memory search (`hybrid`) | Changes only the public Memory search mode. | +| `query-time-compact-v1` | PreparedContext (8,000 bytes) | Keeps ingestion fixed and compacts context at query time through `/v1/context/prepare`. | +| `write-time-l0-l1-v1` | Memory search (`fts`) | Writes bounded deterministic L0 and L1 Memory entries without adding persistent schema fields. | +| `task-lensed-selection-v1` | Memory search (`fts`) | Uses a deterministic question-keyword query projection without label, answer, or Judge access. | + +Unregistered IDs (currently the unsupported temporal-filter arm) are rejected before any stage +runs instead of silently falling back to FTS. `--search-mode` was removed from +`retrieval-smoke` and `run-smoke`; the arm is the single source of the search mode. + +Guarantees implemented and unit-tested: + +- The runner reads `/v1/capabilities` before ingestion and fails as an + `infrastructure` capability error when the Server lacks the arm's search mode or + `powercontext.prepared-context.v1` support. +- The adapter sends the arm's `mode` to the public `/v1/memory/search` endpoint, reads the + executed mode from the response, and records `requested_mode`/`actual_mode` in + `adapter-audit.jsonl`, in each `retrieval-results.jsonl` row, and through `post_query_hook`. +- The query-time compact arm calls the public PreparedContext endpoint with an 8,000-byte + budget, validates the response byte count, and records `retrieval_strategy: prepared-context`. +- Every explicitly requested search mode must match the executed mode the Server reports; + only `auto` accepts the Server's own choice. A mismatch is recorded as an `integrity` + failure, never as a wrong answer or a silent fallback to another mode. +- Every run manifest, retrieval manifest and summary, and unified report records the full arm + block, and `report.md` shows the arm ID. +- `ensure_comparable_experiment_runs` refuses to compare two runs whose dataset lock digest, + question manifest digest, harness commit, processor model or revision, context budget, search + limit, Reader/Judge configuration, or PowerContext/integration revisions differ beyond the + arm, and requires both arm records and every comparison field to be present. + +## Reproducibility and privacy + +- The harness commit, dataset revision and file digests, smoke question order, processor + revision, model settings, PowerContext revision, and integration revision are recorded. +- The approximately 1.2 GB trajectory input remains streaming; it is not loaded as one string or + byte array. +- The adapter, retrieval, preparation, and Reader stages do not read question type, reference + answers, or Judge data. Only local scoring reads the locked reference answers. +- API credentials are resolved from named environment variables. Secret values are not written + to manifests, reports, failure evidence, examples, or Git. +- `scoring-inputs.local.jsonl` contains local replay evidence and must remain outside Git, + shared reports, and telemetry. +- Evaluation artifacts remain outside normal Runtime persistence and outside this repository. + +## Validation evidence + +A real one-command, model-free run completed the preflight, retrieval, and prompt-preparation +stages against a local PowerContext Server: + +| Measurement | Result | +| --- | --- | +| Classification | `smoke-subset` | +| Status | `partial` (Reader, Judge, and replay intentionally skipped) | +| Questions | 10 | +| Failed stages | 0 | +| Retrieved context items | 100 | +| Prepared context tokens | 239,765 | +| Ingestion latency | 121,838.524 ms | +| Published accuracy | unavailable, as required for a model-free run | + +The generated `report.json` and `report.md` contained no credential values. The local server was +stopped after validation, and the run artifacts were not added to Git. + +A separate paid smoke execution of the saved ten Reader inputs produced 4 correct answers out of +10 with the configured DeepSeek Reader/Judge path. This is a smoke-subset observation only. It is +not a full LongMemEval-V2 result and does not establish product or model leadership. The saved +score was also reproduced by the offline replay path with zero Reader and Judge calls. + +The experiment-arm framework was verified on the same local setup. The Server's +`/v1/capabilities` reported `search_modes: ["auto", "fts"]` (no embedding model is configured on +this machine), so hybrid execution could not be exercised locally; per the agreed scope, the +machine-level verification is FTS success plus hybrid capability rejection: + +| Verification | Result | +| --- | --- | +| Model-free FTS arm run (`--experiment-arm current-memory-fts-v1 --skip-reader`) | `preflight`, `retrieval`, `prepare` completed; 10/10 questions succeeded; all ten queries show `requested: "fts"`, `actual: "fts"` in the audit and per-question results. | +| Hybrid arm run (`--experiment-arm current-memory-hybrid-v1 --skip-reader`) | Failed before ingestion: `RetrievalCapabilityError`, error class `infrastructure`, phase `retrieval`; no retrieval artifacts and no ingestion; the manifest and report still record the hybrid arm. | +| Query-time compact run (`--experiment-arm query-time-compact-v1 --skip-reader`) | `preflight`, `retrieval`, and `prepare` completed; 10/10 questions succeeded through public PreparedContext; 10 context items were bounded to 80,000 bytes / 24,623 tokens in total, with no model calls. | + +Arm verification run directories (kept outside the repository under the external runs root): +`smoke-v1-arm-fts-model-free-20260920T191032` and +`smoke-v1-arm-hybrid-capability-check-20260920T191032`, plus +`smoke-v1-query-time-compact-model-free-20260920T2345`. + +The scoped verification for this delivery completed with: + +- 182 LongMemEval-V2 unit tests passing; +- Ruff lint and format checks passing; +- targeted type checks for the arm registry, adapter, retrieval runner, orchestrator, report, + CLI, and their tests passing; +- `git diff --check` passing; and +- secret-pattern scans of changed files and the real model-free runs returning no matches. + +## Explicitly not executed + +No full-tier dataset run was executed. The repository does not yet contain an approved full-tier +dataset lock or question manifest, and hard case/cost guards are not implemented. A paid full run +therefore requires separate implementation review and explicit approval for cost and data egress. +See [longmemeval-v2-full-run.md](longmemeval-v2-full-run.md) for the recorded prerequisites. + +Hybrid retrieval was not executed against a hybrid-capable Server: this machine's Server +advertises only `auto` and `fts` because no embedding model is configured. A temporal-filter +arm is intentionally not registered because the pinned public dataset has ordered states but no +timestamps, and the public Memory search contract exposes no time-range filter. It must fail +loudly rather than treating haystack order as fabricated time provenance. + +This workload does not validate Handoff, cross-host recovery, normal Runtime persistence, or +Work Continuity. It does not replace LoCoMo, SWE-bench Pro, or PowerContext-native acceptance +tests. diff --git a/evaluation/scripts/run_longmemeval_v2_harness.py b/evaluation/scripts/run_longmemeval_v2_harness.py new file mode 100644 index 0000000000..4680b12331 --- /dev/null +++ b/evaluation/scripts/run_longmemeval_v2_harness.py @@ -0,0 +1,29 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Run the pinned LongMemEval-V2 harness with the PowerContext adapter registered.""" + +from __future__ import annotations + +import sys +from importlib import import_module +from pathlib import Path + +EVALUATION_ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(EVALUATION_ROOT / "src")) +main = import_module("powercontext_eval.benchmarks.longmemeval_v2.harness_bootstrap").main + + +if __name__ == "__main__": + main() diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/adapter.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/adapter.py new file mode 100644 index 0000000000..4d789842d7 --- /dev/null +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/adapter.py @@ -0,0 +1,786 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""LongMemEval-V2 Memory adapter backed by the public PowerContext HTTP API.""" + +from __future__ import annotations + +import hashlib +import json +import os +import re +import threading +import time +from collections.abc import Callable, Mapping +from copy import deepcopy +from datetime import UTC, datetime +from ipaddress import ip_address +from pathlib import Path +from typing import Protocol, cast, runtime_checkable +from urllib.error import HTTPError, URLError +from urllib.parse import urlsplit +from urllib.request import Request, urlopen + +AUDIT_SCHEMA = "powercontext.longmemeval-v2-memory-audit.v1" +DEFAULT_MEMORY_KIND = "longmemeval_v2_trajectory" +DEFAULT_SOURCE_CHUNK_BYTES = 180_000 +MAX_SOURCE_CHUNK_BYTES = 190_000 +MAX_MEMORY_TEXT_BYTES = 8_192 +SEARCH_MODES = ("auto", "fts", "vector", "hybrid") +QUERY_STRATEGIES = ("memory-search", "prepared-context") +PREPARED_CONTEXT_SCHEMA = "powercontext.prepared-context.v1" +MEMORY_PROJECTIONS = ("deterministic-compact-v1", "deterministic-l0-l1-v1") +TASK_LENSES = ("question-keywords-v1",) +_TASK_LENS_STOPWORDS = frozenset( + { + "about", + "and", + "after", + "are", + "before", + "could", + "from", + "for", + "have", + "into", + "should", + "that", + "the", + "their", + "there", + "these", + "they", + "this", + "what", + "when", + "where", + "which", + "with", + "would", + "your", + "you", + } +) +_HARNESS_RUNTIME_KEYS = { + "cancel_event", + "generation_temperature", + "generation_top_p", + "query_trace_dir", +} +_AUDIT_LOCKS: dict[Path, threading.Lock] = {} +_AUDIT_LOCKS_GUARD = threading.Lock() + + +class PowerContextMemoryAdapterError(RuntimeError): + """The adapter input, transport, or response violated its contract.""" + + +class PowerContextMemoryModeError(PowerContextMemoryAdapterError): + """The search response did not execute an explicitly required search mode.""" + + +@runtime_checkable +class PowerContextRuntime(Protocol): + """Narrow synchronous facade over supported public PowerContext operations.""" + + def capture_content_source(self, payload: Mapping[str, object]) -> Mapping[str, object]: ... + + def remember_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: ... + + def search_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: ... + + def prepare_context(self, payload: Mapping[str, object]) -> Mapping[str, object]: ... + + +class PowerContextHTTPRuntime: + """Call the public PowerContext API without depending on package internals.""" + + def __init__( + self, + base_url: str, + *, + token: str | None, + timeout_seconds: float, + ) -> None: + if not isinstance(base_url, str) or not base_url.strip(): + raise PowerContextMemoryAdapterError("base_url must be a non-empty string") + _validate_transport(base_url) + if timeout_seconds <= 0: + raise PowerContextMemoryAdapterError("timeout_seconds must be positive") + self._base_url = base_url.rstrip("/") + self._token = token + self._timeout_seconds = timeout_seconds + + def capture_content_source(self, payload: Mapping[str, object]) -> Mapping[str, object]: + return self._post("/v1/sources/content", payload, expected_status=202) + + def remember_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + return self._post("/v1/memory/remember", payload, expected_status=200) + + def search_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + return self._post("/v1/memory/search", payload, expected_status=200) + + def prepare_context(self, payload: Mapping[str, object]) -> Mapping[str, object]: + return self._post("/v1/context/prepare", payload, expected_status=200) + + def create_scope(self, payload: Mapping[str, object]) -> Mapping[str, object]: + return self._post("/v1/scopes", payload, expected_status=201) + + def get_readiness(self) -> Mapping[str, object]: + return self._get("/health/ready", expected_status=200) + + def get_capabilities(self) -> Mapping[str, object]: + return self._get("/v1/capabilities", expected_status=200) + + def _post( + self, + path: str, + payload: Mapping[str, object], + *, + expected_status: int, + ) -> Mapping[str, object]: + headers = {"Content-Type": "application/json"} + if self._token: + headers["Authorization"] = f"Bearer {self._token}" + request = Request( + f"{self._base_url}{path}", + data=json.dumps(payload, ensure_ascii=False, separators=(",", ":")).encode(), + headers=headers, + method="POST", + ) + return self._request(request, path=path, expected_status=expected_status) + + def _get(self, path: str, *, expected_status: int) -> Mapping[str, object]: + headers = {} if self._token is None else {"Authorization": f"Bearer {self._token}"} + request = Request(f"{self._base_url}{path}", headers=headers, method="GET") + return self._request(request, path=path, expected_status=expected_status) + + def _request(self, request: Request, *, path: str, expected_status: int) -> Mapping[str, object]: + try: + with urlopen(request, timeout=self._timeout_seconds) as response: + status = response.status + body = response.read() + except HTTPError as error: + raise PowerContextMemoryAdapterError(f"PowerContext {path} returned HTTP {error.code}") from error + except (OSError, URLError) as error: + raise PowerContextMemoryAdapterError(f"PowerContext {path} request failed") from error + if status != expected_status: + raise PowerContextMemoryAdapterError( + f"PowerContext {path} returned HTTP {status}, expected {expected_status}" + ) + try: + value = json.loads(body) + except (UnicodeDecodeError, json.JSONDecodeError) as error: + raise PowerContextMemoryAdapterError(f"PowerContext {path} returned invalid JSON") from error + if not isinstance(value, dict): + raise PowerContextMemoryAdapterError(f"PowerContext {path} returned a non-object response") + return value + + +class PowerContextMemory: + """Duck-typed implementation of the pinned LongMemEval-V2 ``Memory`` contract. + + ``scope_id`` and ``audit_path`` are required memory parameters. Optional + transport, chunking, and search parameters contain configuration only; the + bearer token is resolved at runtime through ``token_env`` and is never + included in ``memory_config`` or audit records. + """ + + memory_type = "powercontext" + + def __init__(self, memory_params: dict[str, object]) -> None: + self.memory_params = dict(memory_params) + self.scope_id = _nonblank(self.memory_params.get("scope_id"), "scope_id") + self.audit_path = Path(_nonblank(self.memory_params.get("audit_path"), "audit_path")) + self.memory_kind = _optional_nonblank( + self.memory_params.get("memory_kind"), + "memory_kind", + DEFAULT_MEMORY_KIND, + ) + self.search_mode = _optional_nonblank(self.memory_params.get("search_mode"), "search_mode", "auto") + if self.search_mode not in SEARCH_MODES: + raise PowerContextMemoryAdapterError( + f"search_mode must be one of {', '.join(SEARCH_MODES)}, not {self.search_mode!r}" + ) + self.query_strategy = _optional_nonblank( + self.memory_params.get("query_strategy"), "query_strategy", "memory-search" + ) + if self.query_strategy not in QUERY_STRATEGIES: + raise PowerContextMemoryAdapterError( + f"query_strategy must be one of {', '.join(QUERY_STRATEGIES)}, not {self.query_strategy!r}" + ) + self.prepared_context_max_bytes = _integer( + self.memory_params.get("prepared_context_max_bytes", 8_000), + "prepared_context_max_bytes", + minimum=512, + maximum=32_768, + ) + self.memory_projection = _optional_nonblank( + self.memory_params.get("memory_projection"), "memory_projection", "deterministic-compact-v1" + ) + if self.memory_projection not in MEMORY_PROJECTIONS: + raise PowerContextMemoryAdapterError( + f"memory_projection must be one of {', '.join(MEMORY_PROJECTIONS)}, not {self.memory_projection!r}" + ) + task_lens = self.memory_params.get("task_lens") + if task_lens is not None and task_lens not in TASK_LENSES: + raise PowerContextMemoryAdapterError(f"unsupported task_lens: {task_lens!r}") + self.task_lens = cast("str | None", task_lens) + self.search_limit = _integer(self.memory_params.get("search_limit", 10), "search_limit", minimum=1, maximum=50) + self.source_chunk_bytes = _integer( + self.memory_params.get("source_chunk_bytes", DEFAULT_SOURCE_CHUNK_BYTES), + "source_chunk_bytes", + minimum=512, + maximum=MAX_SOURCE_CHUNK_BYTES, + ) + self._runtime: PowerContextRuntime = self._http_runtime() + self._clock_ns: Callable[[], int] = time.perf_counter_ns + self._audit_lock = _audit_lock_for(self.audit_path) + self._insert_lock = threading.Lock() + self._inserted_trajectory_ids: set[str] = set() + self._query_context_local = threading.local() + + @property + def memory_config(self) -> dict[str, object]: + """Return only reconstructable configuration, never runtime credentials.""" + + return {"memory_type": self.memory_type, "memory_params": dict(self.memory_params)} + + def configure_runtime(self, **kwargs: object) -> None: + """Inject a fake/runtime clock after construction without persisting it.""" + + unexpected = set(kwargs) - {"runtime", "clock_ns"} - _HARNESS_RUNTIME_KEYS + if unexpected: + raise PowerContextMemoryAdapterError(f"unsupported runtime overrides: {sorted(unexpected)}") + runtime = kwargs.get("runtime") + if runtime is not None: + if not isinstance(runtime, PowerContextRuntime): + raise PowerContextMemoryAdapterError("runtime does not implement the PowerContext adapter operations") + self._runtime = runtime + clock_ns = kwargs.get("clock_ns") + if clock_ns is not None: + if not callable(clock_ns): + raise PowerContextMemoryAdapterError("clock_ns must be callable") + self._clock_ns = cast("Callable[[], int]", clock_ns) + + def insert(self, trajectory: dict[str, object]) -> None: + """Capture one projected trajectory as paired Source and Memory chunks.""" + + trajectory_id, projected = _project_trajectory(trajectory) + source_chunks = _utf8_chunks(projected, self.source_chunk_bytes) + memory_texts = _memory_projection_texts(projected, self.memory_projection) + digest = f"sha256:{hashlib.sha256(projected.encode()).hexdigest()}" + started = self._clock_ns() + capture_ns = 0 + remember_ns = 0 + source_refs: list[dict[str, object]] = [] + memory_citations: list[dict[str, object]] = [] + status = "failed" + failure: str | None = None + try: + with self._insert_lock: + if trajectory_id in self._inserted_trajectory_ids: + raise PowerContextMemoryAdapterError(f"duplicate trajectory insert: {trajectory_id}") + for index, content in enumerate(source_chunks): + source_id = f"longmemeval-v2-{trajectory_id}-{index + 1:04d}" + stage_started = self._clock_ns() + source = self._runtime.capture_content_source( + { + "scope_id": self.scope_id, + "source_id": source_id, + "content": content, + "metadata": { + "benchmark": "longmemeval-v2", + "trajectory_id": trajectory_id, + "chunk_index": index, + "chunk_count": len(source_chunks), + "trajectory_digest": digest, + }, + } + ) + capture_ns += self._clock_ns() - stage_started + source_refs.append(_source_ref(source)) + + for layer, memory_text in memory_texts: + stage_started = self._clock_ns() + memory = self._runtime.remember_memory( + { + "scope_id": self.scope_id, + "kind": self.memory_kind, + "text": memory_text, + "reason": f"LongMemEval-V2 {layer} memory for trajectory {trajectory_id}", + } + ) + remember_ns += self._clock_ns() - stage_started + memory_citations.append(_memory_citation(memory)) + self._inserted_trajectory_ids.add(trajectory_id) + status = "succeeded" + except Exception as error: + failure = type(error).__name__ + raise + finally: + self._write_audit( + { + "operation": "ingest", + "status": status, + "scope_id": self.scope_id, + "trajectory_id": trajectory_id, + "trajectory_digest": digest, + "memory_projection": self.memory_projection, + "source_chunk_count": len(source_chunks), + "memory_entry_count": len(memory_citations), + "source_refs": source_refs, + "memory_citations": memory_citations, + "timings_ms": { + "source_capture": _milliseconds(capture_ns), + "memory_remember": _milliseconds(remember_ns), + "total": _milliseconds(self._clock_ns() - started), + }, + "failure_type": failure, + } + ) + + def query(self, query: str, query_image: str | None = None) -> list[dict[str, str]]: + """Return Memory search hits as LongMemEval-V2 text context items.""" + + question = _nonblank(query, "query") + retrieval_query = _task_lensed_query(question, self.task_lens) + if query_image is not None and (not isinstance(query_image, str) or not query_image.strip()): + raise PowerContextMemoryAdapterError("query_image must be null or a non-empty string") + started = self._clock_ns() + search_ns = 0 + format_ns = 0 + citations: list[dict[str, object]] = [] + items: list[dict[str, str]] = [] + actual_mode: str | None = None + status = "failed" + failure: str | None = None + requested_mode = self.search_mode if self.query_strategy == "memory-search" else None + try: + stage_started = self._clock_ns() + if self.query_strategy == "prepared-context": + response = self._runtime.prepare_context( + { + "scope_id": self.scope_id, + "query": retrieval_query, + "max_bytes": self.prepared_context_max_bytes, + } + ) + search_ns = self._clock_ns() - stage_started + items.extend(_prepared_context_items(response, maximum_bytes=self.prepared_context_max_bytes)) + else: + response = self._runtime.search_memory( + { + "scope_id": self.scope_id, + "query": retrieval_query, + "limit": self.search_limit, + "mode": self.search_mode, + } + ) + search_ns = self._clock_ns() - stage_started + actual_mode = _response_search_mode(response) + _require_requested_mode(self.search_mode, actual_mode) + raw_hits = response.get("hits") + if not isinstance(raw_hits, list): + raise PowerContextMemoryAdapterError("search response hits must be an array") + for index, hit in enumerate(raw_hits): + if not isinstance(hit, Mapping): + raise PowerContextMemoryAdapterError(f"search hit {index} must be an object") + text = hit.get("text") + if not isinstance(text, str) or not text: + raise PowerContextMemoryAdapterError(f"search hit {index} text must be non-empty") + citation = hit.get("citation") + if not isinstance(citation, Mapping): + raise PowerContextMemoryAdapterError(f"search hit {index} citation must be an object") + citations.append(_string_keyed_mapping(citation, f"search hit {index} citation")) + items.append({"type": "text", "value": text}) + stage_started = self._clock_ns() + format_ns = self._clock_ns() - stage_started + status = "succeeded" + return items + except Exception as error: + failure = type(error).__name__ + raise + finally: + query_invocation_id = self.get_query_context().get("query_invocation_id") + timings = { + "search": _milliseconds(search_ns), + "format": _milliseconds(format_ns), + "total": _milliseconds(self._clock_ns() - started), + } + self._query_context_local.last_query_metadata = { + "query_invocation_id": query_invocation_id, + "result_count": len(items), + "retrieval_strategy": self.query_strategy, + "task_lens": self.task_lens, + "requested_mode": requested_mode, + "actual_mode": actual_mode, + "citations": deepcopy(citations), + "timings_ms": dict(timings), + } + self._write_audit( + { + "operation": "query", + "status": status, + "scope_id": self.scope_id, + "query_invocation_id": query_invocation_id, + "query_sha256": hashlib.sha256(question.encode()).hexdigest(), + "retrieval_query_sha256": hashlib.sha256(retrieval_query.encode()).hexdigest(), + "query_image_present": query_image is not None, + "retrieval_strategy": self.query_strategy, + "task_lens": self.task_lens, + "requested_mode": requested_mode, + "actual_mode": actual_mode, + "result_count": len(items), + "citations": citations, + "timings_ms": timings, + "failure_type": failure, + } + ) + + def set_query_context(self, *, query_invocation_id: str) -> None: + identifier = _nonblank(query_invocation_id, "query_invocation_id") + self._query_context_local.context = {"query_invocation_id": identifier} + + def clear_query_context(self) -> None: + if hasattr(self._query_context_local, "context"): + delattr(self._query_context_local, "context") + if hasattr(self._query_context_local, "last_query_metadata"): + delattr(self._query_context_local, "last_query_metadata") + if hasattr(self._query_context_local, "last_query_metadata"): + delattr(self._query_context_local, "last_query_metadata") + + def get_query_context(self) -> dict[str, str]: + context = getattr(self._query_context_local, "context", None) + return dict(context) if isinstance(context, dict) else {} + + def post_query_hook( + self, + *, + query: str, + query_image: str | None, + memory_context: list[dict[str, str]], + ) -> dict[str, object] | None: + metadata = getattr(self._query_context_local, "last_query_metadata", None) + return deepcopy(metadata) if isinstance(metadata, dict) else None + + def _http_runtime(self) -> PowerContextHTTPRuntime: + base_url = _optional_nonblank(self.memory_params.get("base_url"), "base_url", "http://127.0.0.1:8765") + token_env = _optional_nonblank(self.memory_params.get("token_env"), "token_env", "POWERCONTEXT_TOKEN") + timeout = _number(self.memory_params.get("timeout_seconds", 30.0), "timeout_seconds") + return PowerContextHTTPRuntime(base_url, token=os.getenv(token_env), timeout_seconds=timeout) + + def _write_audit(self, event: dict[str, object]) -> None: + record = {"schema": AUDIT_SCHEMA, "observed_at": datetime.now(UTC).isoformat(), **event} + line = json.dumps(record, ensure_ascii=False, separators=(",", ":"), allow_nan=False) + "\n" + with self._audit_lock: + self.audit_path.parent.mkdir(parents=True, exist_ok=True) + with self.audit_path.open("a", encoding="utf-8", newline="") as stream: + stream.write(line) + + +def _project_trajectory(trajectory: Mapping[str, object]) -> tuple[str, str]: + trajectory_id = _nonblank(trajectory.get("id"), "trajectory.id") + states = trajectory.get("states") + if not isinstance(states, list) or not states: + raise PowerContextMemoryAdapterError(f"trajectory {trajectory_id} states must be a non-empty array") + projected_states: list[dict[str, object]] = [] + for index, raw_state in enumerate(states): + if not isinstance(raw_state, Mapping): + raise PowerContextMemoryAdapterError(f"trajectory {trajectory_id} state {index} must be an object") + projected_states.append( + { + "state_index": raw_state.get("state_index", index), + "step": raw_state.get("step", index), + "url": raw_state.get("url"), + "action": raw_state.get("action"), + "thought": raw_state.get("thought", raw_state.get("thoughts")), + "accessibility_tree": raw_state.get("accessibility_tree", raw_state.get("text")), + "screenshot": raw_state.get("screenshot"), + } + ) + projected = { + "id": trajectory_id, + "domain": trajectory.get("domain"), + "environment": trajectory.get("environment"), + "goal": trajectory.get("goal"), + "outcome": trajectory.get("outcome"), + "start_url": trajectory.get("start_url"), + "states": projected_states, + } + return trajectory_id, json.dumps(projected, ensure_ascii=False, separators=(",", ":"), allow_nan=False) + + +def _utf8_chunks(text: str, maximum_bytes: int) -> list[str]: + encoded = text.encode() + chunks: list[str] = [] + offset = 0 + while offset < len(encoded): + end = min(offset + maximum_bytes, len(encoded)) + while end > offset: + try: + chunk = encoded[offset:end].decode() + break + except UnicodeDecodeError: + end -= 1 + if end == offset: + raise PowerContextMemoryAdapterError("cannot split trajectory into UTF-8 chunks") + chunks.append(chunk) + offset = end + return chunks + + +def _compact_memory_text(projected: str, *, heading: str = "LongMemEval-V2 deterministic trajectory memory") -> str: + value = json.loads(projected) + if not isinstance(value, dict): + raise PowerContextMemoryAdapterError("projected trajectory must be an object") + lines = [ + heading, + f"trajectory_id: {_plain(value.get('id'))}", + f"domain: {_plain(value.get('domain'))}", + f"environment: {_plain(value.get('environment'))}", + f"goal: {_bounded(value.get('goal'), 1_000)}", + f"outcome: {_plain(value.get('outcome'))}", + f"start_url: {_bounded(value.get('start_url'), 500)}", + ] + states = value.get("states") + if isinstance(states, list): + for index, state in enumerate(states): + if not isinstance(state, Mapping): + continue + lines.extend( + [ + f"state {index} url: {_bounded(state.get('url'), 300)}", + f"state {index} action: {_bounded(state.get('action'), 600)}", + f"state {index} thought: {_bounded(state.get('thought'), 600)}", + f"state {index} observation: {_bounded(state.get('accessibility_tree'), 600)}", + ] + ) + return _truncate_utf8_middle("\n".join(lines), MAX_MEMORY_TEXT_BYTES) + + +def _memory_projection_texts(projected: str, projection: str) -> list[tuple[str, str]]: + if projection == "deterministic-compact-v1": + return [("deterministic compact", _compact_memory_text(projected))] + if projection != "deterministic-l0-l1-v1": + raise PowerContextMemoryAdapterError(f"unsupported memory projection: {projection}") + value = json.loads(projected) + if not isinstance(value, dict): + raise PowerContextMemoryAdapterError("projected trajectory must be an object") + l0 = _truncate_utf8_middle( + "\n".join( + [ + "LongMemEval-V2 deterministic L0 trajectory index", + f"trajectory_id: {_plain(value.get('id'))}", + f"domain: {_plain(value.get('domain'))}", + f"environment: {_plain(value.get('environment'))}", + f"goal: {_bounded(value.get('goal'), 700)}", + f"outcome: {_plain(value.get('outcome'))}", + ] + ), + 1_024, + ) + # The final heading is built before truncation so the byte limit always applies + # to the text the Server will actually receive. + l1 = _compact_memory_text(projected, heading="LongMemEval-V2 deterministic L1 trajectory summary") + return [("deterministic L0", l0), ("deterministic L1", l1)] + + +def _task_lensed_query(question: str, task_lens: str | None) -> str: + if task_lens is None: + return question + if task_lens != "question-keywords-v1": + raise PowerContextMemoryAdapterError(f"unsupported task_lens: {task_lens}") + keywords: list[str] = [] + seen: set[str] = set() + for token in re.findall(r"[A-Za-z0-9][A-Za-z0-9_-]{2,}", question.lower()): + if token in _TASK_LENS_STOPWORDS or token in seen: + continue + seen.add(token) + keywords.append(token) + if len(keywords) == 24: + break + return " ".join(keywords) or question + + +def _plain(value: object) -> str: + return value if isinstance(value, str) and value else "" + + +def _bounded(value: object, maximum_chars: int) -> str: + text = _plain(value).replace("\x00", " ") + if len(text) <= maximum_chars: + return text + head = maximum_chars // 2 + tail = maximum_chars - head + return f"{text[:head]}…{text[-tail:]}" + + +def _truncate_utf8_middle(text: str, maximum_bytes: int) -> str: + encoded = text.encode() + if len(encoded) <= maximum_bytes: + return text + marker = "\n… …\n" + marker_bytes = marker.encode() + budget = maximum_bytes - len(marker_bytes) + head = _utf8_prefix(encoded, budget // 2) + tail = _utf8_suffix(encoded, budget - len(head)) + return head.decode() + marker + tail.decode() + + +def _utf8_prefix(value: bytes, maximum_bytes: int) -> bytes: + end = min(len(value), maximum_bytes) + while end > 0: + try: + value[:end].decode() + return value[:end] + except UnicodeDecodeError: + end -= 1 + return b"" + + +def _utf8_suffix(value: bytes, maximum_bytes: int) -> bytes: + start = max(0, len(value) - maximum_bytes) + while start < len(value): + try: + value[start:].decode() + return value[start:] + except UnicodeDecodeError: + start += 1 + return b"" + + +def _source_ref(response: Mapping[str, object]) -> dict[str, object]: + value = response.get("source") + if not isinstance(value, Mapping): + raise PowerContextMemoryAdapterError("capture response source must be an object") + return _string_keyed_mapping(value, "capture response source") + + +def _memory_citation(response: Mapping[str, object]) -> dict[str, object]: + entry = response.get("entry") + if not isinstance(entry, Mapping): + raise PowerContextMemoryAdapterError("remember response entry must be an object") + citation = entry.get("citation") + if not isinstance(citation, Mapping): + raise PowerContextMemoryAdapterError("remember response citation must be an object") + return _string_keyed_mapping(citation, "remember response citation") + + +def _response_search_mode(response: Mapping[str, object]) -> str | None: + """Read the server-reported executed mode; the public contract marks it nullable.""" + + value = response.get("mode") + return value.strip() if isinstance(value, str) and value.strip() else None + + +def _prepared_context_items(response: Mapping[str, object], *, maximum_bytes: int) -> list[dict[str, str]]: + """Validate one public PreparedContext response and render upstream text items.""" + + if response.get("schema") != PREPARED_CONTEXT_SCHEMA: + raise PowerContextMemoryAdapterError("prepared context response has an unsupported schema") + status = response.get("status") + content = response.get("content") + content_bytes = response.get("content_bytes") + if isinstance(content_bytes, bool) or not isinstance(content_bytes, int) or content_bytes < 0: + raise PowerContextMemoryAdapterError("prepared context content_bytes must be a non-negative integer") + if content_bytes > maximum_bytes: + raise PowerContextMemoryAdapterError("prepared context exceeded the requested byte budget") + if status == "empty": + if content is not None or content_bytes != 0: + raise PowerContextMemoryAdapterError("empty prepared context must have null content and zero bytes") + return [] + if status != "ready" or not isinstance(content, str) or not content: + raise PowerContextMemoryAdapterError("ready prepared context must contain non-empty text") + if len(content.encode()) != content_bytes: + raise PowerContextMemoryAdapterError("prepared context content_bytes does not match its UTF-8 content") + return [{"type": "text", "value": content}] + + +def _require_requested_mode(requested_mode: str, actual_mode: str | None) -> None: + """Fail closed when an explicitly requested search mode was not executed. + + ``auto`` delegates the mode choice to the Server and accepts whatever the Server + reports. Every explicit mode — the only modes an experiment arm may request — must + match the reported mode exactly, so an arm's provenance stays provable and two arms + cannot silently execute the same retrieval path. + """ + + if requested_mode != "auto" and actual_mode != requested_mode: + raise PowerContextMemoryModeError( + f"{requested_mode} Memory search was requested but the server reported mode {actual_mode!r}" + ) + + +def _string_keyed_mapping(value: object, label: str) -> dict[str, object]: + if not isinstance(value, Mapping): + raise PowerContextMemoryAdapterError(f"{label} must be an object") + if not all(isinstance(key, str) for key in value): + raise PowerContextMemoryAdapterError(f"{label} keys must be strings") + return {cast(str, key): item for key, item in value.items()} + + +def _nonblank(value: object, label: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise PowerContextMemoryAdapterError(f"{label} must be a non-empty string") + return value.strip() + + +def _optional_nonblank(value: object, label: str, default: str) -> str: + return default if value is None else _nonblank(value, label) + + +def _integer(value: object, label: str, *, minimum: int, maximum: int) -> int: + if isinstance(value, bool) or not isinstance(value, int) or not minimum <= value <= maximum: + raise PowerContextMemoryAdapterError(f"{label} must be an integer from {minimum} through {maximum}") + return value + + +def _number(value: object, label: str) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)) or value <= 0: + raise PowerContextMemoryAdapterError(f"{label} must be positive") + return float(value) + + +def _milliseconds(nanoseconds: int) -> float: + return round(nanoseconds / 1_000_000, 3) + + +def _audit_lock_for(path: Path) -> threading.Lock: + identity = path.resolve() + with _AUDIT_LOCKS_GUARD: + lock = _AUDIT_LOCKS.get(identity) + if lock is None: + lock = threading.Lock() + _AUDIT_LOCKS[identity] = lock + return lock + + +def _validate_transport(base_url: str) -> None: + parsed = urlsplit(base_url) + if parsed.scheme not in {"http", "https"} or not parsed.hostname: + raise PowerContextMemoryAdapterError("base_url must be an absolute HTTP(S) URL") + if parsed.username is not None or parsed.password is not None: + raise PowerContextMemoryAdapterError("base_url must not contain credentials") + if parsed.query or parsed.fragment: + raise PowerContextMemoryAdapterError("base_url must not contain query or fragment data") + if parsed.scheme == "https": + return + hostname = parsed.hostname + try: + is_loopback = ip_address(hostname).is_loopback + except ValueError: + is_loopback = hostname.lower() == "localhost" + if not is_loopback: + raise PowerContextMemoryAdapterError("refusing unencrypted non-loopback PowerContext transport") diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/arms.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/arms.py new file mode 100644 index 0000000000..791a7867c0 --- /dev/null +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/arms.py @@ -0,0 +1,222 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Experiment arm registry for comparable LongMemEval-V2 retrieval runs.""" + +from __future__ import annotations + +from collections.abc import Mapping +from dataclasses import dataclass +from typing import Any + +from powercontext_eval.errors import PowerContextEvalError + + +class ExperimentArmError(PowerContextEvalError): + """An experiment arm lookup or run comparison cannot preserve its contract.""" + + +@dataclass(frozen=True) +class ExperimentArm: + """One pinned experiment configuration identity for the smoke workload. + + Two runs are comparable when every field outside this record matches, so an arm + must declare every behavioural knob it is allowed to vary. Only arms whose + configuration is actually implemented may be registered; unimplemented arms must + fail loudly instead of silently executing the default behaviour. + """ + + arm_id: str + retrieval_strategy: str + search_mode: str | None + memory_projection: str + query_projection: str + prepared_context_max_bytes: int | None + temporal_filter: str | None + task_lens: str | None + + +CURRENT_MEMORY_FTS = ExperimentArm( + arm_id="current-memory-fts-v1", + retrieval_strategy="memory-search", + search_mode="fts", + memory_projection="deterministic-compact-v1", + query_projection="question-text-v1", + prepared_context_max_bytes=None, + temporal_filter=None, + task_lens=None, +) + +CURRENT_MEMORY_HYBRID = ExperimentArm( + arm_id="current-memory-hybrid-v1", + retrieval_strategy="memory-search", + search_mode="hybrid", + memory_projection="deterministic-compact-v1", + query_projection="question-text-v1", + prepared_context_max_bytes=None, + temporal_filter=None, + task_lens=None, +) + +QUERY_TIME_COMPACT = ExperimentArm( + arm_id="query-time-compact-v1", + retrieval_strategy="prepared-context", + search_mode=None, + memory_projection="deterministic-compact-v1", + query_projection="powercontext-prepared-context-v1", + prepared_context_max_bytes=8_000, + temporal_filter=None, + task_lens=None, +) + +WRITE_TIME_L0_L1 = ExperimentArm( + arm_id="write-time-l0-l1-v1", + retrieval_strategy="memory-search", + search_mode="fts", + memory_projection="deterministic-l0-l1-v1", + query_projection="question-text-v1", + prepared_context_max_bytes=None, + temporal_filter=None, + task_lens=None, +) + +TASK_LENSED_SELECTION = ExperimentArm( + arm_id="task-lensed-selection-v1", + retrieval_strategy="memory-search", + search_mode="fts", + memory_projection="deterministic-compact-v1", + query_projection="deterministic-keyword-lens-v1", + prepared_context_max_bytes=None, + temporal_filter=None, + task_lens="question-keywords-v1", +) + +EXPERIMENT_ARMS: tuple[ExperimentArm, ...] = ( + CURRENT_MEMORY_FTS, + CURRENT_MEMORY_HYBRID, + QUERY_TIME_COMPACT, + WRITE_TIME_L0_L1, + TASK_LENSED_SELECTION, +) +DEFAULT_EXPERIMENT_ARM_ID = CURRENT_MEMORY_FTS.arm_id + +_ARMS_BY_ID = {arm.arm_id: arm for arm in EXPERIMENT_ARMS} + +# Run-manifest fields that must match before two runs may be compared. Everything +# else — run identity, timestamps, machine-local paths, and the arm record itself — +# may differ. ``powercontext.search_mode`` is excluded because it is derived from +# the arm and is exactly the declared difference between two arms. +_COMPARED_MANIFEST_PATHS: tuple[tuple[str, ...], ...] = ( + ("inputs", "dataset_lock", "content_sha256"), + ("inputs", "smoke_manifest", "content_sha256"), + ("inputs", "harness", "commit"), + ("processor", "model"), + ("processor", "revision"), + ("memory_context_max_tokens",), + ("powercontext", "search_limit"), + ("reader",), + ("judge",), + ("revisions", "powercontext"), + ("revisions", "integration"), +) + +# ``reader`` and ``judge`` are legitimately null for model-free runs, so a null value +# there is a comparable configuration rather than a missing field. +_NULLABLE_COMPARED_PATHS = frozenset({("reader",), ("judge",)}) + + +def get_experiment_arm(arm_id: str) -> ExperimentArm: + """Return the registered arm for ``arm_id`` or fail with the supported list.""" + + arm = _ARMS_BY_ID.get(arm_id) if isinstance(arm_id, str) else None + if arm is None: + supported = ", ".join(registered.arm_id for registered in EXPERIMENT_ARMS) + raise ExperimentArmError(f"unknown experiment arm: {arm_id!r}; supported arms: {supported}") + return arm + + +def supported_experiment_arm_ids() -> tuple[str, ...]: + """Return the stable IDs of every registered experiment arm.""" + + return tuple(arm.arm_id for arm in EXPERIMENT_ARMS) + + +def resolve_experiment_arm(value: str | ExperimentArm | None) -> ExperimentArm: + """Resolve a CLI argument or direct value into one experiment arm. + + ``None`` keeps the default FTS arm so existing callers keep their behaviour. + """ + + if isinstance(value, ExperimentArm): + return value + if value is None: + return CURRENT_MEMORY_FTS + return get_experiment_arm(value) + + +def arm_manifest_record(arm: ExperimentArm) -> dict[str, object]: + """Render the arm as the stable block recorded in every run manifest and report.""" + + return { + "id": arm.arm_id, + "retrieval_strategy": arm.retrieval_strategy, + "search_mode": arm.search_mode, + "memory_projection": arm.memory_projection, + "query_projection": arm.query_projection, + "prepared_context_max_bytes": arm.prepared_context_max_bytes, + "temporal_filter": arm.temporal_filter, + "task_lens": arm.task_lens, + } + + +def ensure_comparable_experiment_runs(manifest_a: Mapping[str, object], manifest_b: Mapping[str, object]) -> None: + """Refuse to compare two runs whose pinned conditions differ beyond the arm. + + Every fair-comparison field must be present in both manifests, both + ``experiment_arm`` records must be registered arm configurations, and only the arm + record and run-identity fields may differ. + """ + + differences: list[str] = [] + differences.extend(_unregistered_arm_differences(manifest_a, "first manifest")) + differences.extend(_unregistered_arm_differences(manifest_b, "second manifest")) + for path in _COMPARED_MANIFEST_PATHS: + name = ".".join(path) + value_a = _manifest_value(manifest_a, path) + value_b = _manifest_value(manifest_b, path) + if path not in _NULLABLE_COMPARED_PATHS and (value_a is None or value_b is None): + differences.append(f"{name} is missing") + elif value_a != value_b: + differences.append(name) + if differences: + raise ExperimentArmError("runs are not comparable: " + "; ".join(differences)) + + +def _unregistered_arm_differences(manifest: Mapping[str, object], label: str) -> list[str]: + value = manifest.get("experiment_arm") + if not isinstance(value, Mapping): + return [f"{label} has no experiment_arm record"] + record = {str(key): item for key, item in value.items()} + if any(record == arm_manifest_record(arm) for arm in EXPERIMENT_ARMS): + return [] + return [f"{label} has an unregistered experiment_arm record"] + + +def _manifest_value(manifest: Mapping[str, object], path: tuple[str, ...]) -> Any: + value: object = manifest + for key in path: + if not isinstance(value, Mapping): + return None + value = value.get(key) + return value diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/catalog.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/catalog.py index b190345ac3..2e59b14275 100644 --- a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/catalog.py +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/catalog.py @@ -23,7 +23,7 @@ from dataclasses import dataclass from pathlib import Path from types import MappingProxyType -from typing import Literal, TypeAlias +from typing import Literal, TypeAlias, cast from powercontext_eval.errors import PowerContextEvalError @@ -32,6 +32,7 @@ SMOKE_MANIFEST_SCHEMA = "powercontext.longmemeval-v2-smoke.v1" DATASET_LOCK_SCHEMA = "powercontext.longmemeval-v2-dataset-lock.v1" RUN_INPUT_MANIFEST_SCHEMA = "powercontext.longmemeval-v2-run-input.v1" +RUN_SUBSET_SCHEMA = "powercontext.longmemeval-v2-run-subset.v1" Tier: TypeAlias = Literal["small", "medium"] Ability: TypeAlias = Literal[ @@ -68,6 +69,14 @@ class LongMemEvalV2CatalogError(PowerContextEvalError): """The pinned LongMemEval-V2 inputs cannot be trusted.""" +class LongMemEvalV2InputError(LongMemEvalV2CatalogError): + """The requested smoke configuration is invalid.""" + + +class LongMemEvalV2EnvironmentError(LongMemEvalV2CatalogError): + """The pinned data, harness, or output environment is not ready.""" + + @dataclass(frozen=True) class Question: """The query-visible fields required to validate one upstream question.""" @@ -124,6 +133,16 @@ def as_json(self) -> dict[str, object]: "cases": [{"question_id": case.question_id, "ability": case.ability} for case in self.cases], } + def as_run_artifact_json(self) -> dict[str, object]: + """Identify one validated selection as a generated smoke artifact.""" + + return { + "schema": RUN_SUBSET_SCHEMA, + "classification": "smoke-subset", + "tier": self.tier, + "cases": [{"question_id": case.question_id, "ability": case.ability} for case in self.cases], + } + @dataclass(frozen=True) class LongMemEvalV2Catalog: @@ -146,7 +165,7 @@ def load( """Load and validate the three upstream files required by one tier.""" if tier not in {"small", "medium"}: - raise LongMemEvalV2CatalogError(f"Unsupported LongMemEval-V2 tier: {tier}") + raise LongMemEvalV2InputError(f"Unsupported LongMemEval-V2 tier: {tier}") root = data_root.resolve() paths = { "questions.jsonl": root / "questions.jsonl", @@ -172,19 +191,19 @@ def select_smoke(self, cases: Sequence[SmokeCase]) -> SmokeSelection: """Validate a fixed smoke subset without deriving one from benchmark metadata.""" if not cases: - raise LongMemEvalV2CatalogError("Smoke subset must contain at least one question") + raise LongMemEvalV2InputError("Smoke subset must contain at least one question") selected: list[SmokeCase] = [] seen: set[str] = set() source_positions = {question_id: index for index, question_id in enumerate(self.source_question_ids)} for case in cases: if case.question_id in seen: - raise LongMemEvalV2CatalogError(f"Smoke subset contains duplicate question id: {case.question_id}") + raise LongMemEvalV2InputError(f"Smoke subset contains duplicate question id: {case.question_id}") seen.add(case.question_id) question = self.questions.get(case.question_id) if question is None: - raise LongMemEvalV2CatalogError(f"Smoke subset references an unknown question id: {case.question_id}") + raise LongMemEvalV2InputError(f"Smoke subset references an unknown question id: {case.question_id}") if question.ability != case.ability: - raise LongMemEvalV2CatalogError( + raise LongMemEvalV2InputError( f"Smoke subset ability mismatch for {case.question_id}: " f"expected {question.ability}, got {case.ability}" ) @@ -192,11 +211,11 @@ def select_smoke(self, cases: Sequence[SmokeCase]) -> SmokeSelection: if [source_positions[case.question_id] for case in selected] != sorted( source_positions[case.question_id] for case in selected ): - raise LongMemEvalV2CatalogError("Smoke subset question order must match the upstream source order") + raise LongMemEvalV2InputError("Smoke subset question order must match the upstream source order") selected_abilities = {case.ability for case in selected} missing = sorted(_ABILITIES - selected_abilities) if missing: - raise LongMemEvalV2CatalogError(f"Smoke subset is missing published abilities: {', '.join(missing)}") + raise LongMemEvalV2InputError(f"Smoke subset is missing published abilities: {', '.join(missing)}") return SmokeSelection(tier=self.tier, cases=tuple(selected)) @@ -204,30 +223,36 @@ def load_smoke_manifest(path: Path) -> SmokeSelection: """Load one exact smoke-subset declaration without selecting questions dynamically.""" try: - value = json.loads(path.read_text(encoding="utf-8")) - except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error: - raise LongMemEvalV2CatalogError(f"Cannot read LongMemEval-V2 smoke manifest: {path}") from error + content = path.read_text(encoding="utf-8") + except OSError as error: + raise LongMemEvalV2EnvironmentError(f"Cannot read LongMemEval-V2 smoke manifest: {path}") from error + except UnicodeDecodeError as error: + raise LongMemEvalV2InputError("LongMemEval-V2 smoke manifest is not valid UTF-8 JSON") from error + try: + value = json.loads(content) + except json.JSONDecodeError as error: + raise LongMemEvalV2InputError("LongMemEval-V2 smoke manifest is not valid UTF-8 JSON") from error if not isinstance(value, dict) or set(value) != {"schema", "tier", "cases"}: - raise LongMemEvalV2CatalogError("Smoke manifest must contain only schema, tier, and cases") + raise LongMemEvalV2InputError("Smoke manifest must contain only schema, tier, and cases") if value["schema"] != SMOKE_MANIFEST_SCHEMA: - raise LongMemEvalV2CatalogError("Smoke manifest schema is unsupported") + raise LongMemEvalV2InputError("Smoke manifest schema is unsupported") tier = value["tier"] if tier not in {"small", "medium"}: - raise LongMemEvalV2CatalogError("Smoke manifest tier must be small or medium") + raise LongMemEvalV2InputError("Smoke manifest tier must be small or medium") raw_cases = value["cases"] if not isinstance(raw_cases, list): - raise LongMemEvalV2CatalogError("Smoke manifest cases must be an array") + raise LongMemEvalV2InputError("Smoke manifest cases must be an array") cases: list[SmokeCase] = [] for index, raw_case in enumerate(raw_cases): if not isinstance(raw_case, dict) or set(raw_case) != {"question_id", "ability"}: - raise LongMemEvalV2CatalogError(f"Smoke manifest case {index} has an invalid shape") - question_id = raw_case["question_id"] - ability = raw_case["ability"] + raise LongMemEvalV2InputError(f"Smoke manifest case {index} has an invalid shape") + question_id = raw_case.get("question_id") + ability = raw_case.get("ability") if not isinstance(question_id, str) or not question_id.strip(): - raise LongMemEvalV2CatalogError(f"Smoke manifest case {index} has an invalid question id") + raise LongMemEvalV2InputError(f"Smoke manifest case {index} has an invalid question id") if ability not in _ABILITIES: - raise LongMemEvalV2CatalogError(f"Smoke manifest case {index} has an invalid ability") - cases.append(SmokeCase(question_id=question_id, ability=ability)) + raise LongMemEvalV2InputError(f"Smoke manifest case {index} has an invalid ability") + cases.append(SmokeCase(question_id=question_id, ability=cast(Ability, ability))) return SmokeSelection(tier=tier, cases=tuple(cases)) @@ -235,26 +260,32 @@ def load_dataset_lock(path: Path) -> DatasetLock: """Load a fixed input identity before reading any benchmark data.""" try: - value = json.loads(path.read_text(encoding="utf-8")) - except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error: - raise LongMemEvalV2CatalogError(f"Cannot read LongMemEval-V2 dataset lock: {path}") from error + content = path.read_text(encoding="utf-8") + except OSError as error: + raise LongMemEvalV2EnvironmentError(f"Cannot read LongMemEval-V2 dataset lock: {path}") from error + except UnicodeDecodeError as error: + raise LongMemEvalV2InputError("LongMemEval-V2 dataset lock is not valid UTF-8 JSON") from error + try: + value = json.loads(content) + except json.JSONDecodeError as error: + raise LongMemEvalV2InputError("LongMemEval-V2 dataset lock is not valid UTF-8 JSON") from error if not isinstance(value, dict) or set(value) != {"schema", "upstream", "dataset_revision", "tier", "files"}: - raise LongMemEvalV2CatalogError( + raise LongMemEvalV2InputError( "Dataset lock must contain only schema, upstream, dataset_revision, tier, and files" ) if value["schema"] != DATASET_LOCK_SCHEMA: - raise LongMemEvalV2CatalogError("Dataset lock schema is unsupported") + raise LongMemEvalV2InputError("Dataset lock schema is unsupported") upstream = value["upstream"] if not isinstance(upstream, dict) or set(upstream) != {"repository", "harness_commit"}: - raise LongMemEvalV2CatalogError("Dataset lock upstream identity is invalid") + raise LongMemEvalV2InputError("Dataset lock upstream identity is invalid") if upstream["repository"] != UPSTREAM_REPOSITORY or upstream["harness_commit"] != UPSTREAM_HARNESS_COMMIT: - raise LongMemEvalV2CatalogError("Dataset lock does not match the pinned LongMemEval-V2 harness") + raise LongMemEvalV2InputError("Dataset lock does not match the pinned LongMemEval-V2 harness") dataset_revision = value["dataset_revision"] if not isinstance(dataset_revision, str) or not dataset_revision.strip(): - raise LongMemEvalV2CatalogError("Dataset lock dataset revision must be non-empty") + raise LongMemEvalV2InputError("Dataset lock dataset revision must be non-empty") tier = value["tier"] if tier not in {"small", "medium"}: - raise LongMemEvalV2CatalogError("Dataset lock tier must be small or medium") + raise LongMemEvalV2InputError("Dataset lock tier must be small or medium") files = value["files"] expected_names = { "questions.jsonl", @@ -262,9 +293,9 @@ def load_dataset_lock(path: Path) -> DatasetLock: f"haystacks/lme_v2_{tier}.json", } if not isinstance(files, dict) or set(files) != expected_names: - raise LongMemEvalV2CatalogError("Dataset lock must provide digests for exactly the required files") + raise LongMemEvalV2InputError("Dataset lock must provide digests for exactly the required files") if any(not isinstance(digest, str) or len(digest) != 64 or not _is_lower_hex(digest) for digest in files.values()): - raise LongMemEvalV2CatalogError("Dataset lock file digests must be lowercase SHA-256 values") + raise LongMemEvalV2InputError("Dataset lock file digests must be lowercase SHA-256 values") return DatasetLock(tier=tier, dataset_revision=dataset_revision, file_digests=MappingProxyType(dict(files))) @@ -273,14 +304,14 @@ def validate_harness_checkout(harness_root: Path) -> None: harness = harness_root.resolve() if not (harness / "evaluation" / "harness.py").is_file(): - raise LongMemEvalV2CatalogError(f"LongMemEval-V2 harness is missing evaluation/harness.py: {harness}") + raise LongMemEvalV2EnvironmentError(f"LongMemEval-V2 harness is missing evaluation/harness.py: {harness}") revision = _harness_git(harness, "rev-parse", "HEAD").stdout.strip() if revision != UPSTREAM_HARNESS_COMMIT: - raise LongMemEvalV2CatalogError( + raise LongMemEvalV2EnvironmentError( f"LongMemEval-V2 harness checkout must be {UPSTREAM_HARNESS_COMMIT}, got {revision or 'unknown'}" ) if _harness_git(harness, "diff", "--quiet", "HEAD", "--").returncode != 0: - raise LongMemEvalV2CatalogError("LongMemEval-V2 harness checkout has tracked changes relative to HEAD") + raise LongMemEvalV2EnvironmentError("LongMemEval-V2 harness checkout has tracked changes relative to HEAD") def _harness_git(harness: Path, *arguments: str) -> subprocess.CompletedProcess[str]: @@ -295,7 +326,7 @@ def _harness_git(harness: Path, *arguments: str) -> subprocess.CompletedProcess[ timeout=30, ) except (OSError, subprocess.SubprocessError) as error: - raise LongMemEvalV2CatalogError(f"Cannot inspect LongMemEval-V2 harness checkout: {harness}") from error + raise LongMemEvalV2EnvironmentError(f"Cannot inspect LongMemEval-V2 harness checkout: {harness}") from error def _file_digest(path: Path, label: str) -> str: @@ -309,9 +340,9 @@ def _file_digest(path: Path, label: str) -> str: hasher.update(chunk) bytes_read += len(chunk) except OSError as error: - raise LongMemEvalV2CatalogError(f"Missing LongMemEval-V2 input {label}: {path}") from error + raise LongMemEvalV2EnvironmentError(f"Missing LongMemEval-V2 input {label}: {path}") from error if bytes_read == 0: - raise LongMemEvalV2CatalogError(f"LongMemEval-V2 input {label} is blank") + raise LongMemEvalV2EnvironmentError(f"LongMemEval-V2 input {label} is blank") return hasher.hexdigest() @@ -319,10 +350,10 @@ def _validate_expected_digests(actual: Mapping[str, str], expected: Mapping[str, if expected is None: return if set(expected) != set(actual): - raise LongMemEvalV2CatalogError("Expected input digests must name exactly the required files") + raise LongMemEvalV2InputError("Expected input digests must name exactly the required files") for name, digest in actual.items(): if expected[name] != digest: - raise LongMemEvalV2CatalogError(f"LongMemEval-V2 SHA-256 mismatch for {name}") + raise LongMemEvalV2EnvironmentError(f"LongMemEval-V2 SHA-256 mismatch for {name}") def _questions(path: Path) -> tuple[dict[str, Question], tuple[str, ...]]: @@ -331,15 +362,19 @@ def _questions(path: Path) -> tuple[dict[str, Question], tuple[str, ...]]: for index, row in enumerate(_jsonl(path, "questions.jsonl")): question_id = _nonblank(row.get("id"), f"question {index} id") if question_id in questions: - raise LongMemEvalV2CatalogError(f"Duplicate LongMemEval-V2 question id: {question_id}") + raise LongMemEvalV2EnvironmentError(f"Duplicate LongMemEval-V2 question id: {question_id}") domain = row.get("domain") if domain not in {"web", "enterprise"}: - raise LongMemEvalV2CatalogError(f"Invalid question domain for {question_id}") + raise LongMemEvalV2EnvironmentError(f"Invalid question domain for {question_id}") _nonblank(row.get("question"), f"question text for {question_id}") question_type = row.get("question_type") if question_type not in _QUESTION_TYPE_ABILITIES: - raise LongMemEvalV2CatalogError(f"Unsupported question type for {question_id}") - questions[question_id] = Question(question_id=question_id, domain=domain, question_type=question_type) + raise LongMemEvalV2EnvironmentError(f"Unsupported question type for {question_id}") + questions[question_id] = Question( + question_id=question_id, + domain=cast(Literal["web", "enterprise"], domain), + question_type=cast(str, question_type), + ) source_order.append(question_id) return questions, tuple(source_order) @@ -349,11 +384,11 @@ def _trajectories(path: Path) -> Mapping[str, Literal["web", "enterprise"]]: for index, row in enumerate(_jsonl(path, "trajectories.jsonl")): trajectory_id = _nonblank(row.get("id"), f"trajectory {index} id") if trajectory_id in trajectories: - raise LongMemEvalV2CatalogError(f"Duplicate LongMemEval-V2 trajectory id: {trajectory_id}") + raise LongMemEvalV2EnvironmentError(f"Duplicate LongMemEval-V2 trajectory id: {trajectory_id}") domain = row.get("domain") if domain not in {"web", "enterprise"}: - raise LongMemEvalV2CatalogError(f"Invalid trajectory domain for {trajectory_id}") - trajectories[trajectory_id] = domain + raise LongMemEvalV2EnvironmentError(f"Invalid trajectory domain for {trajectory_id}") + trajectories[trajectory_id] = cast(Literal["web", "enterprise"], domain) return MappingProxyType(trajectories) @@ -362,21 +397,21 @@ def _haystack(path: Path) -> Mapping[str, tuple[str, ...]]: with path.open(encoding="utf-8") as source: value = json.load(source) except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error: - raise LongMemEvalV2CatalogError("LongMemEval-V2 haystack is not valid JSON") from error + raise LongMemEvalV2EnvironmentError("LongMemEval-V2 haystack is not valid JSON") from error if not isinstance(value, dict): - raise LongMemEvalV2CatalogError("LongMemEval-V2 haystack must be a JSON object") + raise LongMemEvalV2EnvironmentError("LongMemEval-V2 haystack must be a JSON object") haystack: dict[str, tuple[str, ...]] = {} for question_id, trajectory_ids in value.items(): if not isinstance(question_id, str) or not question_id: - raise LongMemEvalV2CatalogError("LongMemEval-V2 haystack contains an invalid question id") + raise LongMemEvalV2EnvironmentError("LongMemEval-V2 haystack contains an invalid question id") if not isinstance(trajectory_ids, list) or not trajectory_ids: - raise LongMemEvalV2CatalogError(f"LongMemEval-V2 haystack is empty for {question_id}") + raise LongMemEvalV2EnvironmentError(f"LongMemEval-V2 haystack is empty for {question_id}") if not all(isinstance(trajectory_id, str) and trajectory_id for trajectory_id in trajectory_ids): - raise LongMemEvalV2CatalogError( + raise LongMemEvalV2EnvironmentError( f"LongMemEval-V2 haystack contains an invalid trajectory id for {question_id}" ) if len(trajectory_ids) != len(set(trajectory_ids)): - raise LongMemEvalV2CatalogError( + raise LongMemEvalV2EnvironmentError( f"LongMemEval-V2 haystack contains duplicate trajectories for {question_id}" ) haystack[question_id] = tuple(trajectory_ids) @@ -389,17 +424,19 @@ def _validate_haystack( haystack: Mapping[str, tuple[str, ...]], ) -> None: if set(haystack) != set(questions): - raise LongMemEvalV2CatalogError("LongMemEval-V2 questions and haystack ids must match exactly") + raise LongMemEvalV2EnvironmentError("LongMemEval-V2 questions and haystack ids must match exactly") for question_id, trajectory_ids in haystack.items(): question = questions[question_id] for trajectory_id in trajectory_ids: trajectory_domain = trajectories.get(trajectory_id) if trajectory_domain is None: - raise LongMemEvalV2CatalogError( + raise LongMemEvalV2EnvironmentError( f"LongMemEval-V2 haystack references an unknown trajectory: {trajectory_id}" ) if trajectory_domain != question.domain: - raise LongMemEvalV2CatalogError(f"LongMemEval-V2 haystack crosses domains for question {question_id}") + raise LongMemEvalV2EnvironmentError( + f"LongMemEval-V2 haystack crosses domains for question {question_id}" + ) def _jsonl(path: Path, label: str) -> Iterator[dict[str, object]]: @@ -415,23 +452,23 @@ def _jsonl(path: Path, label: str) -> Iterator[dict[str, object]]: try: value = json.loads(line) except json.JSONDecodeError as error: - raise LongMemEvalV2CatalogError( + raise LongMemEvalV2EnvironmentError( f"LongMemEval-V2 input {label} has invalid JSON at {line_number}" ) from error if not isinstance(value, dict): - raise LongMemEvalV2CatalogError( + raise LongMemEvalV2EnvironmentError( f"LongMemEval-V2 input {label} has a non-object row at {line_number}" ) yield value except (OSError, UnicodeDecodeError) as error: - raise LongMemEvalV2CatalogError(f"Cannot read LongMemEval-V2 input {label}: {path}") from error + raise LongMemEvalV2EnvironmentError(f"Cannot read LongMemEval-V2 input {label}: {path}") from error if not found: - raise LongMemEvalV2CatalogError(f"LongMemEval-V2 input {label} is blank") + raise LongMemEvalV2EnvironmentError(f"LongMemEval-V2 input {label} is blank") def _nonblank(value: object, label: str) -> str: if not isinstance(value, str) or not value.strip(): - raise LongMemEvalV2CatalogError(f"Invalid LongMemEval-V2 {label}") + raise LongMemEvalV2EnvironmentError(f"Invalid LongMemEval-V2 {label}") return value diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/costs.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/costs.py new file mode 100644 index 0000000000..548562e868 --- /dev/null +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/costs.py @@ -0,0 +1,303 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Explicit price policies and usage-based cost records for LongMemEval-V2 model stages. + +Prices are never hardcoded: every price comes from an explicitly provided policy, so a +recorded cost always names the provider, model, currency, per-million prices, and the +policy revision it was computed under. + +A cost is computed only from real provider usage, and only when that usage reports the +cache split the policy prices. Cached input and uncached input have different prices (the +DeepSeek cache-hit rate is roughly fifty times cheaper than the cache-miss rate), so a +total-input-token approximation would misprice long, highly repeated prompts. Whenever the +split is missing, the provider/model does not match the policy, or no policy is configured, +the cost stays ``null`` with a reason — never ``0``, because a missing price is not a free +model and an unpriced token class is not free either. +""" + +from __future__ import annotations + +import math +from collections.abc import Mapping +from dataclasses import dataclass + +from powercontext_eval.errors import PowerContextEvalError + +# The amount fields are named ``*_usd`` and are summed across stages, so only a policy that +# actually prices United States dollars may produce them. +SUPPORTED_CURRENCY = "USD" + + +class CostPolicyError(PowerContextEvalError): + """An explicitly provided cost policy is missing, malformed, or unusable.""" + + +@dataclass(frozen=True) +class ModelPricePolicy: + """One explicitly configured price identity for one provider and one model.""" + + provider: str + model: str + currency: str + input_cache_hit_price_per_million: float + input_cache_miss_price_per_million: float + output_price_per_million: float + price_policy_revision: str + + +@dataclass +class UsageAccount: + """Sum per-call model usage so completed calls stay priced even when processing fails. + + ``account`` accepts the normalized per-call usage records the transports produce + (``input_tokens``/``output_tokens`` plus the optional cache split); anything a call + did not report coerces to zero or marks the summed split incomplete. + """ + + calls: int = 0 + input_tokens: int = 0 + output_tokens: int = 0 + cache_hit_tokens: int = 0 + cache_miss_tokens: int = 0 + cache_split_reported: bool = True + + def account(self, usage: object) -> None: + self.calls += 1 + if not isinstance(usage, Mapping): + return + self.input_tokens += _usage_int(usage.get("input_tokens")) + self.output_tokens += _usage_int(usage.get("output_tokens")) + hit = _usage_optional_int(usage.get("input_cache_hit_tokens")) + miss = _usage_optional_int(usage.get("input_cache_miss_tokens")) + if hit is None or miss is None: + self.cache_split_reported = False + else: + self.cache_hit_tokens += hit + self.cache_miss_tokens += miss + + +def parse_cost_policy(value: object, *, label: str = "cost policy") -> ModelPricePolicy | None: + """Parse one explicitly provided price policy; ``None`` means "no prices configured".""" + + if value is None: + return None + if not isinstance(value, Mapping): + raise CostPolicyError(f"{label} must be a JSON object when provided") + required = { + "provider", + "model", + "currency", + "input_cache_hit_price_per_million", + "input_cache_miss_price_per_million", + "output_price_per_million", + "price_policy_revision", + } + if set(value) != required: + raise CostPolicyError(f"{label} must contain exactly {', '.join(sorted(required))}") + record = {str(key): item for key, item in value.items()} + currency = _nonblank(record["currency"], f"{label} currency") + if currency.upper() != SUPPORTED_CURRENCY: + raise CostPolicyError( + f"{label} currency must be {SUPPORTED_CURRENCY}; the recorded amount fields are USD-denominated" + ) + return ModelPricePolicy( + provider=_nonblank(record["provider"], f"{label} provider"), + model=_nonblank(record["model"], f"{label} model"), + currency=SUPPORTED_CURRENCY, + input_cache_hit_price_per_million=_price( + record["input_cache_hit_price_per_million"], f"{label} input_cache_hit_price_per_million" + ), + input_cache_miss_price_per_million=_price( + record["input_cache_miss_price_per_million"], f"{label} input_cache_miss_price_per_million" + ), + output_price_per_million=_price(record["output_price_per_million"], f"{label} output_price_per_million"), + price_policy_revision=_nonblank(record["price_policy_revision"], f"{label} price_policy_revision"), + ) + + +def cost_policy_record(policy: ModelPricePolicy | None) -> dict[str, object] | None: + """Render a policy as the identity block recorded in manifests and summaries.""" + + if policy is None: + return None + return { + "provider": policy.provider, + "model": policy.model, + "currency": policy.currency, + "input_cache_hit_price_per_million": policy.input_cache_hit_price_per_million, + "input_cache_miss_price_per_million": policy.input_cache_miss_price_per_million, + "output_price_per_million": policy.output_price_per_million, + "price_policy_revision": policy.price_policy_revision, + } + + +def usage_cost_usd( + policy: ModelPricePolicy, + *, + cache_hit_tokens: int, + cache_miss_tokens: int, + output_tokens: int, +) -> float: + """Price real provider-reported usage, charging cached and uncached input separately.""" + + cache_hit_tokens = _token_count(cache_hit_tokens, "cache_hit_tokens") + cache_miss_tokens = _token_count(cache_miss_tokens, "cache_miss_tokens") + output_tokens = _token_count(output_tokens, "output_tokens") + cost = (cache_hit_tokens / 1_000_000) * policy.input_cache_hit_price_per_million + cost += (cache_miss_tokens / 1_000_000) * policy.input_cache_miss_price_per_million + cost += (output_tokens / 1_000_000) * policy.output_price_per_million + return round(cost, 6) + + +def usage_cost_block( + policy: ModelPricePolicy | None, + *, + provider: str, + model: str, + input_tokens: int, + cache_hit_tokens: int | None, + cache_miss_tokens: int | None, + output_tokens: int, +) -> dict[str, object]: + """Render one stage's real usage with its cost, or a null cost and the unavailable reason. + + The cost requires a policy that matches both the provider and the model, plus a usage + record that reports the cache hit/miss split the policy prices. Anything else records + ``cost_usd: null`` with the reason instead of a number. + """ + + provider = _nonblank(provider, "provider") + model = _nonblank(model, "model") + input_tokens = _token_count(input_tokens, "input_tokens") + output_tokens = _token_count(output_tokens, "output_tokens") + hit = _optional_token_count(cache_hit_tokens, "cache_hit_tokens") + miss = _optional_token_count(cache_miss_tokens, "cache_miss_tokens") + usage = { + "provider": provider, + "model": model, + "input_tokens": input_tokens, + "input_cache_hit_tokens": hit, + "input_cache_miss_tokens": miss, + "output_tokens": output_tokens, + } + reason = _unavailable_reason( + policy, + provider=provider, + model=model, + input_tokens=input_tokens, + hit=hit, + miss=miss, + ) + if reason is not None: + return _unpriced_block(usage, reason) + assert policy is not None and hit is not None and miss is not None # narrowed by _unavailable_reason + return { + **usage, + "cost_usd": usage_cost_usd(policy, cache_hit_tokens=hit, cache_miss_tokens=miss, output_tokens=output_tokens), + "currency": policy.currency, + "input_cache_hit_price_per_million": policy.input_cache_hit_price_per_million, + "input_cache_miss_price_per_million": policy.input_cache_miss_price_per_million, + "output_price_per_million": policy.output_price_per_million, + "price_policy_revision": policy.price_policy_revision, + "unavailable_reason": None, + } + + +def model_free_cost_block(*, stage: str, reason: str) -> dict[str, object]: + """Render the zero-cost record for a stage that provably called no model.""" + + return { + "stage": _nonblank(stage, "stage"), + "input_tokens": 0, + "input_cache_hit_tokens": 0, + "input_cache_miss_tokens": 0, + "output_tokens": 0, + "cost_usd": 0.0, + "reason": _nonblank(reason, "reason"), + } + + +def _unavailable_reason( + policy: ModelPricePolicy | None, + *, + provider: str, + model: str, + input_tokens: int, + hit: int | None, + miss: int | None, +) -> str | None: + if policy is None: + return "no price policy was configured for this run" + if policy.provider != provider or policy.model != model: + return ( + f"the configured price policy prices provider {policy.provider!r} model {policy.model!r}, " + f"not provider {provider!r} model {model!r}" + ) + if (hit is None) != (miss is None): + return "provider usage reported only one side of the required cache hit/miss split" + if hit is None: + return "provider usage did not report a cache hit/miss split, so cached input cannot be priced" + assert miss is not None + if hit + miss != input_tokens: + return "provider usage cache hit/miss tokens do not equal reported input tokens" + return None + + +def _unpriced_block(usage: Mapping[str, object], reason: str) -> dict[str, object]: + return { + **usage, + "cost_usd": None, + "currency": None, + "input_cache_hit_price_per_million": None, + "input_cache_miss_price_per_million": None, + "output_price_per_million": None, + "price_policy_revision": None, + "unavailable_reason": reason, + } + + +def _price(value: object, label: str) -> float: + if isinstance(value, bool) or not isinstance(value, (int, float)) or not math.isfinite(float(value)): + raise CostPolicyError(f"{label} must be a finite number") + price = float(value) + if price < 0: + raise CostPolicyError(f"{label} must not be negative") + return price + + +def _token_count(value: int, label: str) -> int: + if isinstance(value, bool) or not isinstance(value, int) or value < 0: + raise CostPolicyError(f"{label} must be a non-negative integer") + return value + + +def _optional_token_count(value: int | None, label: str) -> int | None: + return None if value is None else _token_count(value, label) + + +def _nonblank(value: object, label: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise CostPolicyError(f"{label} must be a non-empty string") + return value.strip() + + +def _usage_int(value: object) -> int: + return value if isinstance(value, int) and not isinstance(value, bool) and value >= 0 else 0 + + +def _usage_optional_int(value: object) -> int | None: + if isinstance(value, int) and not isinstance(value, bool) and value >= 0: + return value + return None diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/harness_bootstrap.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/harness_bootstrap.py new file mode 100644 index 0000000000..828117f72f --- /dev/null +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/harness_bootstrap.py @@ -0,0 +1,116 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Register PowerContext and execute the pinned LongMemEval-V2 harness unchanged.""" + +from __future__ import annotations + +import argparse +import importlib +import runpy +import sys +from collections.abc import Sequence +from contextlib import contextmanager +from pathlib import Path +from types import ModuleType +from typing import Any + +from powercontext_eval.benchmarks.longmemeval_v2.adapter import PowerContextMemory +from powercontext_eval.benchmarks.longmemeval_v2.catalog import validate_harness_checkout + + +class LongMemEvalV2HarnessBootstrapError(RuntimeError): + """The pinned harness could not be safely registered or executed.""" + + +def run_pinned_harness(harness_root: Path, arguments: Sequence[str]) -> None: + """Validate, register the adapter, and run the upstream harness in-process.""" + + root = harness_root.resolve() + validate_harness_checkout(root) + harness_path = root / "evaluation" / "harness.py" + memory_path = root / "memory_modules" / "memory.py" + with _harness_imports(root): + memory_module = importlib.import_module("memory_modules.memory") + loaded_path = Path(_module_path(memory_module)).resolve() + if loaded_path != memory_path: + raise LongMemEvalV2HarnessBootstrapError( + f"loaded LongMemEval-V2 memory contract from {loaded_path}, expected {memory_path}" + ) + registry = getattr(memory_module, "MEMORY_TYPES", None) + register_memory = getattr(memory_module, "register_memory", None) + if not isinstance(registry, dict) or not callable(register_memory): + raise LongMemEvalV2HarnessBootstrapError("LongMemEval-V2 memory registry contract is unavailable") + existing = registry.get(PowerContextMemory.memory_type) + if existing not in {None, PowerContextMemory}: + raise LongMemEvalV2HarnessBootstrapError( + "LongMemEval-V2 already registered a different powercontext backend" + ) + register_memory(PowerContextMemory) + with _arguments(harness_path, arguments): + runpy.run_path(str(harness_path), run_name="__main__") + + +def main(argv: Sequence[str] | None = None) -> None: + """CLI entry point intended to be run with the pinned harness Python.""" + + parser = argparse.ArgumentParser(add_help=False) + parser.add_argument("--harness-root", type=Path, required=True) + known, forwarded = parser.parse_known_args(argv) + if forwarded[:1] == ["--"]: + forwarded = forwarded[1:] + run_pinned_harness(known.harness_root, forwarded) + + +def _module_path(module: ModuleType) -> str: + value = getattr(module, "__file__", None) + if not isinstance(value, str): + raise LongMemEvalV2HarnessBootstrapError("LongMemEval-V2 memory contract has no file identity") + return value + + +@contextmanager +def _arguments(harness_path: Path, arguments: Sequence[str]) -> Any: + previous = sys.argv + sys.argv = [str(harness_path), *arguments] + try: + yield + finally: + sys.argv = previous + + +@contextmanager +def _harness_imports(harness_root: Path) -> Any: + root = str(harness_root) + previous_path = list(sys.path) + previous_modules = { + name: module + for name, module in sys.modules.items() + if name == "memory_modules" or name.startswith("memory_modules.") + } + for name in previous_modules: + del sys.modules[name] + sys.path.insert(0, root) + try: + yield + finally: + sys.path[:] = previous_path + for name in tuple(sys.modules): + if name == "memory_modules" or name.startswith("memory_modules."): + del sys.modules[name] + sys.modules.update(previous_modules) + + +if __name__ == "__main__": + main() diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/prepare_smoke.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/prepare_smoke.py new file mode 100644 index 0000000000..bbf90d8d53 --- /dev/null +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/prepare_smoke.py @@ -0,0 +1,239 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Prepare pinned LongMemEval-V2 Reader prompts without calling a Reader.""" + +from __future__ import annotations + +import hashlib +import json +import os +import subprocess +from dataclasses import dataclass +from pathlib import Path + +from powercontext_eval.benchmarks.longmemeval_v2.catalog import UPSTREAM_HARNESS_COMMIT, validate_harness_checkout +from powercontext_eval.errors import PowerContextEvalError + +DEFAULT_PROCESSOR_MODEL = "Qwen/Qwen3.5-9B" +PREPARE_MANIFEST_SCHEMA = "powercontext.longmemeval-v2-prepare-run.v1" +PREPARED_PROMPT_SCHEMA = "powercontext.longmemeval-v2-prepared-prompt.v1" +PREPARE_FAILURE_SCHEMA = "powercontext.longmemeval-v2-prepare-failure.v1" +PREPARE_SUMMARY_SCHEMA = "powercontext.longmemeval-v2-prepare-summary.v1" + + +class PrepareSmokeError(PowerContextEvalError): + """The retrieval artifacts or pinned harness cannot produce Reader inputs.""" + + +@dataclass(frozen=True) +class PreparedPromptRun: + """Inspectable artifacts emitted before a Reader call.""" + + output_dir: Path + manifest_path: Path + prompts_path: Path + failures_path: Path + summary_path: Path + + +def prepare_reader_inputs_smoke( + *, + retrieval_dir: Path, + harness_root: Path, + harness_python: Path, + output_dir: Path, + processor_revision: str, + processor_model: str = DEFAULT_PROCESSOR_MODEL, + memory_context_max_tokens: int = 200_000, +) -> PreparedPromptRun: + """Use the pinned harness to truncate context and build deterministic Reader messages.""" + + caller_cwd = Path.cwd() + retrieval_dir = _resolve_path(retrieval_dir, caller_cwd) + harness_root = _resolve_path(harness_root, caller_cwd) + harness_python = _resolve_path(harness_python, caller_cwd) + output_dir = _resolve_path(output_dir, caller_cwd) + if output_dir.exists(): + raise PrepareSmokeError(f"Refusing to overwrite prepared prompt artifacts: {output_dir}") + revision = _nonblank(processor_revision, "processor_revision") + model = _nonblank(processor_model, "processor_model") + if isinstance(memory_context_max_tokens, bool) or memory_context_max_tokens <= 0: + raise PrepareSmokeError("memory_context_max_tokens must be positive") + if not harness_python.is_file(): + raise PrepareSmokeError(f"LongMemEval-V2 harness Python does not exist: {harness_python}") + validate_harness_checkout(harness_root) + + retrieval_manifest = _load_json(retrieval_dir / "retrieval-manifest.json", "retrieval manifest") + retrieval_summary = _load_json(retrieval_dir / "summary.json", "retrieval summary") + retrieval_results = retrieval_dir / "retrieval-results.jsonl" + if not retrieval_results.is_file(): + raise PrepareSmokeError(f"Missing retrieval results: {retrieval_results}") + _validate_retrieval_artifacts(retrieval_manifest, retrieval_summary) + + try: + output_dir.mkdir(parents=True, exist_ok=False) + except FileExistsError as error: + raise PrepareSmokeError(f"Refusing to overwrite prepared prompt artifacts: {output_dir}") from error + except OSError as error: + raise PrepareSmokeError(f"Cannot create prepared prompt artifact directory: {output_dir}") from error + + manifest_path = output_dir / "prepare-manifest.json" + prompts_path = output_dir / "prepared-prompts.jsonl" + failures_path = output_dir / "prepare-failures.jsonl" + summary_path = output_dir / "prepare-summary.json" + _write_json_exclusive( + manifest_path, + { + "schema": PREPARE_MANIFEST_SCHEMA, + "classification": "smoke-subset-prepare-only", + "retrieval": { + "directory": str(retrieval_dir.resolve()), + "manifest_sha256": _file_digest(retrieval_dir / "retrieval-manifest.json"), + "results_sha256": _file_digest(retrieval_results), + "summary_sha256": _file_digest(retrieval_dir / "summary.json"), + }, + "harness": { + "commit": UPSTREAM_HARNESS_COMMIT, + "root": str(harness_root.resolve()), + "python": str(harness_python), + }, + "processor": {"model": model, "revision": revision}, + "memory_context_max_tokens": memory_context_max_tokens, + "reader": None, + "judge": None, + }, + ) + + command = [ + str(harness_python), + "-m", + "powercontext_eval.benchmarks.longmemeval_v2.prepare_worker", + "--harness-root", + str(harness_root), + "--input-path", + str(retrieval_results), + "--prompts-path", + str(prompts_path), + "--failures-path", + str(failures_path), + "--summary-path", + str(summary_path), + "--processor-model", + model, + "--processor-revision", + revision, + "--memory-context-max-tokens", + str(memory_context_max_tokens), + ] + environment = dict(os.environ) + source_root = Path(__file__).parents[3] + environment["PYTHONPATH"] = _prepend_path(source_root, environment.get("PYTHONPATH")) + environment["PYTHONNOUSERSITE"] = "1" + try: + completed = subprocess.run( + command, + cwd=harness_root, + env=environment, + capture_output=True, + text=True, + encoding="utf-8", + errors="replace", + check=False, + timeout=600, + ) + except (OSError, subprocess.SubprocessError) as error: + raise PrepareSmokeError("Cannot start pinned LongMemEval-V2 prompt preparation worker") from error + _write_text_exclusive(output_dir / "prepare-worker.stdout.log", completed.stdout) + _write_text_exclusive(output_dir / "prepare-worker.stderr.log", completed.stderr) + if completed.returncode != 0: + raise PrepareSmokeError( + f"Pinned LongMemEval-V2 prompt preparation worker failed with exit {completed.returncode}" + ) + for path in (prompts_path, failures_path, summary_path): + if not path.is_file(): + raise PrepareSmokeError(f"Prompt preparation worker did not produce {path.name}") + return PreparedPromptRun( + output_dir=output_dir, + manifest_path=manifest_path, + prompts_path=prompts_path, + failures_path=failures_path, + summary_path=summary_path, + ) + + +def _validate_retrieval_artifacts(manifest: dict[str, object], summary: dict[str, object]) -> None: + if manifest.get("classification") != "smoke-subset-retrieval-only": + raise PrepareSmokeError("Retrieval manifest is not a retrieval-only smoke subset") + runtime = manifest.get("runtime") + if not isinstance(runtime, dict) or runtime.get("reader") is not None or runtime.get("judge") is not None: + raise PrepareSmokeError("Retrieval artifacts must not contain Reader or Judge execution") + if summary.get("classification") != "smoke-subset-retrieval-only": + raise PrepareSmokeError("Retrieval summary is not a retrieval-only smoke subset") + question_count = summary.get("question_count") + failed = summary.get("failed") + if question_count != 10 or failed != 0: + raise PrepareSmokeError("Retrieval artifacts must contain ten successful smoke questions") + + +def _resolve_path(path: Path, caller_cwd: Path) -> Path: + """Absolutize against the caller's cwd without dereferencing symlinks. + + ``Path.resolve()`` would replace a virtualenv's interpreter with the base Python + it links to, losing the venv's installed dependencies at worker launch. + """ + return Path(os.path.abspath(caller_cwd / path)) + + +def _load_json(path: Path, label: str) -> dict[str, object]: + try: + value = json.loads(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error: + raise PrepareSmokeError(f"Cannot read {label}: {path}") from error + if not isinstance(value, dict): + raise PrepareSmokeError(f"{label} must be a JSON object") + return value + + +def _file_digest(path: Path) -> str: + hasher = hashlib.sha256() + try: + with path.open("rb") as stream: + while chunk := stream.read(1024 * 1024): + hasher.update(chunk) + except OSError as error: + raise PrepareSmokeError(f"Cannot hash prompt preparation input: {path}") from error + return hasher.hexdigest() + + +def _prepend_path(path: Path, current: str | None) -> str: + return str(path) if not current else f"{path}{os.pathsep}{current}" + + +def _write_json_exclusive(path: Path, value: object) -> None: + _write_text_exclusive(path, json.dumps(value, ensure_ascii=True, indent=2, sort_keys=True) + "\n") + + +def _write_text_exclusive(path: Path, value: str) -> None: + try: + with path.open("x", encoding="utf-8", newline="") as stream: + stream.write(value) + except OSError as error: + raise PrepareSmokeError(f"Cannot write prepared prompt artifact: {path}") from error + + +def _nonblank(value: str, label: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise PrepareSmokeError(f"{label} must be a non-empty string") + return value.strip() diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/prepare_worker.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/prepare_worker.py new file mode 100644 index 0000000000..3da4b6ee0a --- /dev/null +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/prepare_worker.py @@ -0,0 +1,265 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Pinned-harness worker that builds bounded LongMemEval-V2 Reader inputs.""" + +from __future__ import annotations + +import argparse +import hashlib +import importlib +import json +import sys +import time +from collections.abc import Iterator, Mapping +from datetime import UTC, datetime +from pathlib import Path +from typing import Any, Protocol, cast + +from powercontext_eval.benchmarks.longmemeval_v2.prepare_smoke import ( + PREPARE_FAILURE_SCHEMA, + PREPARE_SUMMARY_SCHEMA, + PREPARED_PROMPT_SCHEMA, +) + + +class HarnessPromptAPI(Protocol): + def get_system_prompt(self, domain: str) -> str: ... + + def validate_memory_context_items(self, memory_context: Any, *, question_id: str) -> list[dict[str, str]]: ... + + def truncate_memory_context( + self, + memory_context: list[dict[str, str]], + *, + max_tokens: int, + question_id: str, + ) -> tuple[list[dict[str, str]], int, int]: ... + + def build_messages( + self, + *, + system_prompt: str, + question_text: str, + image_path: str | None, + memory_context: list[dict[str, str]], + ) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: ... + + +def main() -> None: + parser = argparse.ArgumentParser(description="Prepare bounded LongMemEval-V2 Reader prompts.") + parser.add_argument("--harness-root", type=Path, required=True) + parser.add_argument("--input-path", type=Path, required=True) + parser.add_argument("--prompts-path", type=Path, required=True) + parser.add_argument("--failures-path", type=Path, required=True) + parser.add_argument("--summary-path", type=Path, required=True) + parser.add_argument("--processor-model", required=True) + parser.add_argument("--processor-revision", required=True) + parser.add_argument("--memory-context-max-tokens", type=int, required=True) + args = parser.parse_args() + if args.memory_context_max_tokens <= 0: + raise SystemExit("memory_context_max_tokens must be positive") + harness = _load_harness(args.harness_root, args.processor_model, args.processor_revision) + _run( + harness, + input_path=args.input_path, + prompts_path=args.prompts_path, + failures_path=args.failures_path, + summary_path=args.summary_path, + processor_model=args.processor_model, + processor_revision=args.processor_revision, + memory_context_max_tokens=args.memory_context_max_tokens, + ) + + +def _load_harness(harness_root: Path, processor_model: str, processor_revision: str) -> HarnessPromptAPI: + root = harness_root.resolve() + if str(root) not in sys.path: + sys.path.insert(0, str(root)) + harness = importlib.import_module("evaluation.harness") + transformers = importlib.import_module("transformers") + processor_class = transformers.AutoProcessor + processor = processor_class.from_pretrained(processor_model, revision=processor_revision) + harness.MEMORY_CONTEXT_PROCESSOR_LOCAL.processor = processor + return cast(HarnessPromptAPI, harness) + + +def _run( + harness: HarnessPromptAPI, + *, + input_path: Path, + prompts_path: Path, + failures_path: Path, + summary_path: Path, + processor_model: str, + processor_revision: str, + memory_context_max_tokens: int, +) -> None: + started_ns = time.perf_counter_ns() + succeeded = 0 + failed = 0 + original_tokens_total = 0 + prepared_tokens_total = 0 + _create_empty(prompts_path) + _create_empty(failures_path) + for sequence, result in enumerate(_jsonl(input_path), start=1): + question_id = _nonblank(result.get("question_id"), "question_id") + try: + prepared = _prepare_record( + harness, + result, + sequence=sequence, + memory_context_max_tokens=memory_context_max_tokens, + ) + except Exception as error: # noqa: BLE001 - one malformed result must not hide later preparation failures + failed += 1 + _append_json( + failures_path, + { + "schema": PREPARE_FAILURE_SCHEMA, + "question_id": question_id, + "phase": "prompt_prepare", + "error_type": type(error).__name__, + "summary": (str(error).strip() or type(error).__name__)[:500], + }, + ) + continue + _append_json(prompts_path, prepared) + succeeded += 1 + original_tokens = prepared["memory_context_original_tokens"] + prepared_tokens = prepared["memory_context_tokens"] + if not isinstance(original_tokens, int) or not isinstance(prepared_tokens, int): + raise TypeError("prepared prompt token counts must be integers") + original_tokens_total += original_tokens + prepared_tokens_total += prepared_tokens + _write_json( + summary_path, + { + "schema": PREPARE_SUMMARY_SCHEMA, + "classification": "smoke-subset-prepare-only", + "completed_at": datetime.now(UTC).isoformat(), + "question_count": succeeded + failed, + "succeeded": succeeded, + "failed": failed, + "memory_context_original_tokens": original_tokens_total, + "memory_context_tokens": prepared_tokens_total, + "memory_context_max_tokens": memory_context_max_tokens, + "processor": {"model": processor_model, "revision": processor_revision}, + "reader": None, + "judge": None, + "elapsed_ms": round((time.perf_counter_ns() - started_ns) / 1_000_000, 3), + }, + ) + if failed: + raise SystemExit(1) + + +def _prepare_record( + harness: HarnessPromptAPI, + result: Mapping[str, object], + *, + sequence: int, + memory_context_max_tokens: int, +) -> dict[str, object]: + started_ns = time.perf_counter_ns() + question_id = _nonblank(result.get("question_id"), "question_id") + domain = _domain(result.get("domain")) + question = result.get("question") + if not isinstance(question, Mapping): + raise TypeError("question must be an object") + question_text = _nonblank(question.get("text"), "question.text") + image_path = question.get("image") + if image_path is not None and (not isinstance(image_path, str) or not image_path.strip()): + raise ValueError("question.image must be null or a non-empty string") + memory_context = harness.validate_memory_context_items(result.get("memory_context"), question_id=question_id) + bounded_context, original_tokens, bounded_tokens = harness.truncate_memory_context( + memory_context, + max_tokens=memory_context_max_tokens, + question_id=question_id, + ) + system_prompt = harness.get_system_prompt(domain) + messages, prompt_messages = harness.build_messages( + system_prompt=system_prompt, + question_text=question_text, + image_path=image_path, + memory_context=bounded_context, + ) + return { + "schema": PREPARED_PROMPT_SCHEMA, + "sequence": sequence, + "question_id": question_id, + "domain": domain, + "question": {"text": question_text, "image": image_path}, + "scope_id": result.get("scope_id"), + "haystack_digest": result.get("haystack_digest"), + "memory_context": bounded_context, + "memory_context_original_tokens": original_tokens, + "memory_context_tokens": bounded_tokens, + "memory_context_max_tokens": memory_context_max_tokens, + "memory_context_was_truncated": original_tokens > bounded_tokens, + "memory_context_bytes": sum(len(item["value"].encode()) for item in bounded_context), + "system_prompt": system_prompt, + "messages": messages, + "prompt_messages": prompt_messages, + "prompt_sha256": _canonical_digest(messages), + "prepare_latency_ms": round((time.perf_counter_ns() - started_ns) / 1_000_000, 3), + } + + +def _jsonl(path: Path) -> Iterator[dict[str, object]]: + with path.open(encoding="utf-8") as stream: + for line_number, line in enumerate(stream, start=1): + try: + value = json.loads(line) + except json.JSONDecodeError as error: + raise ValueError(f"retrieval result {line_number} is not valid JSON") from error + if not isinstance(value, dict): + raise TypeError(f"retrieval result {line_number} is not an object") + yield value + + +def _domain(value: object) -> str: + if value not in {"web", "enterprise"}: + raise ValueError("domain must be web or enterprise") + return cast(str, value) + + +def _nonblank(value: object, label: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ValueError(f"{label} must be a non-empty string") + return value.strip() + + +def _canonical_digest(value: object) -> str: + encoded = json.dumps(value, ensure_ascii=False, sort_keys=True, separators=(",", ":")).encode() + return hashlib.sha256(encoded).hexdigest() + + +def _create_empty(path: Path) -> None: + with path.open("x", encoding="utf-8"): + pass + + +def _append_json(path: Path, value: object) -> None: + with path.open("a", encoding="utf-8", newline="") as stream: + stream.write(json.dumps(value, ensure_ascii=False, separators=(",", ":"), allow_nan=False) + "\n") + + +def _write_json(path: Path, value: object) -> None: + with path.open("x", encoding="utf-8", newline="") as stream: + stream.write(json.dumps(value, ensure_ascii=True, indent=2, sort_keys=True, allow_nan=False) + "\n") + + +if __name__ == "__main__": + main() diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/reader_smoke.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/reader_smoke.py new file mode 100644 index 0000000000..c513b49f30 --- /dev/null +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/reader_smoke.py @@ -0,0 +1,680 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Run one configured Anthropic-compatible Reader over prepared LongMemEval-V2 prompts.""" + +from __future__ import annotations + +import hashlib +import json +import os +import time +from collections.abc import Iterator, Mapping +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path +from typing import Any, Protocol +from urllib.error import HTTPError, URLError +from urllib.parse import urlsplit +from urllib.request import Request, urlopen + +from powercontext_eval.benchmarks.longmemeval_v2.costs import ( + ModelPricePolicy, + UsageAccount, + cost_policy_record, + usage_cost_block, +) +from powercontext_eval.errors import PowerContextEvalError + +DEFAULT_ANTHROPIC_BASE_URL_ENV = "ANTHROPIC_BASE_URL" +DEFAULT_ANTHROPIC_TOKEN_ENV = "ANTHROPIC_AUTH_TOKEN" +DEFAULT_READER_MODEL = "deepseek-flash-latest" +DEFAULT_DEEPSEEK_BASE_URL = "https://api.deepseek.com" +DEFAULT_DEEPSEEK_TOKEN_ENV = "DEEPSEEK_API_KEY" +DEFAULT_DEEPSEEK_MODEL = "deepseek-flash" +READER_MANIFEST_SCHEMA = "powercontext.longmemeval-v2-reader-run.v1" +READER_OUTPUT_SCHEMA = "powercontext.longmemeval-v2-reader-output.v1" +READER_FAILURE_SCHEMA = "powercontext.longmemeval-v2-reader-failure.v1" +READER_SUMMARY_SCHEMA = "powercontext.longmemeval-v2-reader-summary.v1" + + +class ReaderSmokeError(PowerContextEvalError): + """Prepared prompts cannot be sent safely to the configured Reader.""" + + +class ReaderResponseError(ReaderSmokeError): + """A completed Reader response could not be turned into usable text; its usage is preserved.""" + + def __init__(self, message: str, *, usage: dict[str, object], latency_ms: float | None = None) -> None: + super().__init__(message) + self.usage = usage + self.latency_ms = latency_ms + + +class ReaderTransport(Protocol): + """Minimal synchronous Anthropic-compatible Reader transport.""" + + def complete(self, *, system: str, content: list[dict[str, object]]) -> Mapping[str, object]: ... + + +@dataclass(frozen=True) +class ReaderSmokeRun: + """Inspectable artifacts emitted after Reader calls and before scoring.""" + + output_dir: Path + manifest_path: Path + outputs_path: Path + failures_path: Path + summary_path: Path + + +class AnthropicCompatibleReader: + """Use bearer authentication without persisting the token or response headers.""" + + def __init__( + self, + base_url: str, + *, + token: str, + model: str, + max_tokens: int, + temperature: float, + timeout_seconds: float, + ) -> None: + parsed = urlsplit(base_url) + if ( + parsed.scheme != "https" + or not parsed.hostname + or parsed.username + or parsed.password + or parsed.query + or parsed.fragment + ): + raise ReaderSmokeError("Reader base URL must be an HTTPS URL without credentials") + if not token.strip(): + raise ReaderSmokeError("Reader token is empty") + if not model.strip(): + raise ReaderSmokeError("Reader model is empty") + if max_tokens <= 0: + raise ReaderSmokeError("Reader max_tokens must be positive") + if not 0 <= temperature <= 2: + raise ReaderSmokeError("Reader temperature must be from 0 through 2") + if timeout_seconds <= 0: + raise ReaderSmokeError("Reader timeout_seconds must be positive") + self._endpoint = f"{base_url.rstrip('/')}/v1/messages" + self._token = token + self._model = model + self._max_tokens = max_tokens + self._temperature = temperature + self._timeout_seconds = timeout_seconds + + def complete(self, *, system: str, content: list[dict[str, object]]) -> Mapping[str, object]: + request = Request( + self._endpoint, + data=json.dumps( + { + "model": self._model, + "max_tokens": self._max_tokens, + "temperature": self._temperature, + "system": system, + "messages": [{"role": "user", "content": _anthropic_content(content)}], + }, + ensure_ascii=False, + separators=(",", ":"), + ).encode(), + headers={ + "Authorization": f"Bearer {self._token}", + "Content-Type": "application/json", + "anthropic-version": "2023-06-01", + }, + method="POST", + ) + try: + with urlopen(request, timeout=self._timeout_seconds) as response: + body = response.read() + except HTTPError as error: + raise ReaderSmokeError(f"Reader request returned HTTP {error.code}") from error + except (OSError, URLError) as error: + raise ReaderSmokeError("Reader request failed") from error + try: + value = json.loads(body) + except (UnicodeDecodeError, json.JSONDecodeError) as error: + raise ReaderSmokeError("Reader returned invalid JSON") from error + if not isinstance(value, dict): + raise ReaderSmokeError("Reader returned a non-object response") + return value + + +class DeepSeekOpenAIReader: + """Use DeepSeek's OpenAI-compatible Chat Completions API without persisting credentials.""" + + def __init__( + self, + base_url: str, + *, + token: str, + model: str, + max_tokens: int, + temperature: float, + timeout_seconds: float, + ) -> None: + parsed = urlsplit(base_url) + if ( + parsed.scheme != "https" + or not parsed.hostname + or parsed.username + or parsed.password + or parsed.query + or parsed.fragment + ): + raise ReaderSmokeError("Reader base URL must be an HTTPS URL without credentials") + if not token.strip(): + raise ReaderSmokeError("Reader token is empty") + if not model.strip(): + raise ReaderSmokeError("Reader model is empty") + if max_tokens <= 0: + raise ReaderSmokeError("Reader max_tokens must be positive") + if not 0 <= temperature <= 2: + raise ReaderSmokeError("Reader temperature must be from 0 through 2") + if timeout_seconds <= 0: + raise ReaderSmokeError("Reader timeout_seconds must be positive") + self._endpoint = f"{base_url.rstrip('/')}/chat/completions" + self._token = token + self._model = model + self._max_tokens = max_tokens + self._temperature = temperature + self._timeout_seconds = timeout_seconds + + def complete(self, *, system: str, content: list[dict[str, object]]) -> Mapping[str, object]: + request = Request( + self._endpoint, + data=json.dumps( + { + "model": self._model, + "max_tokens": self._max_tokens, + "temperature": self._temperature, + "thinking": {"type": "disabled"}, + "messages": [ + {"role": "system", "content": system}, + {"role": "user", "content": content}, + ], + }, + ensure_ascii=False, + separators=(",", ":"), + ).encode(), + headers={"Authorization": f"Bearer {self._token}", "Content-Type": "application/json"}, + method="POST", + ) + try: + with urlopen(request, timeout=self._timeout_seconds) as response: + body = response.read() + except HTTPError as error: + raise ReaderSmokeError(f"Reader request returned HTTP {error.code}") from error + except (OSError, URLError) as error: + raise ReaderSmokeError("Reader request failed") from error + try: + value = json.loads(body) + except (UnicodeDecodeError, json.JSONDecodeError) as error: + raise ReaderSmokeError("Reader returned invalid JSON") from error + return _normalize_deepseek_response(value) + + +def run_reader_smoke( + *, + prepared_dir: Path, + output_dir: Path, + provider: str = "anthropic-compatible", + base_url: str | None = None, + base_url_env: str = DEFAULT_ANTHROPIC_BASE_URL_ENV, + token_env: str | None = None, + model: str | None = None, + max_tokens: int = 512, + temperature: float = 0.0, + timeout_seconds: float = 120.0, + max_questions: int | None = None, + price_policy: ModelPricePolicy | None = None, + transport: ReaderTransport | None = None, +) -> ReaderSmokeRun: + """Call a Reader sequentially and write answer/usage artifacts without persisting credentials.""" + + if output_dir.exists(): + raise ReaderSmokeError(f"Refusing to overwrite Reader artifacts: {output_dir}") + if provider not in {"anthropic-compatible", "deepseek-openai"}: + raise ReaderSmokeError("provider must be anthropic-compatible or deepseek-openai") + normalized_model = _nonblank( + model or (DEFAULT_DEEPSEEK_MODEL if provider == "deepseek-openai" else DEFAULT_READER_MODEL), "model" + ) + if max_questions is not None and (isinstance(max_questions, bool) or max_questions <= 0): + raise ReaderSmokeError("max_questions must be null or positive") + prepared_manifest = _load_json(prepared_dir / "prepare-manifest.json", "prepare manifest") + prepared_summary = _load_json(prepared_dir / "prepare-summary.json", "prepare summary") + prompts_path = prepared_dir / "prepared-prompts.jsonl" + if not prompts_path.is_file(): + raise ReaderSmokeError(f"Missing prepared prompts: {prompts_path}") + _validate_prepared_artifacts(prepared_manifest, prepared_summary) + + if transport is None: + resolved_token_env = token_env or ( + DEFAULT_DEEPSEEK_TOKEN_ENV if provider == "deepseek-openai" else DEFAULT_ANTHROPIC_TOKEN_ENV + ) + resolved_base_url = _nonblank( + base_url or (DEFAULT_DEEPSEEK_BASE_URL if provider == "deepseek-openai" else os.getenv(base_url_env)), + "base_url", + ) + token = _nonblank(os.getenv(resolved_token_env), resolved_token_env) + if provider == "deepseek-openai": + active_transport = DeepSeekOpenAIReader( + resolved_base_url, + token=token, + model=normalized_model, + max_tokens=max_tokens, + temperature=temperature, + timeout_seconds=timeout_seconds, + ) + else: + active_transport = AnthropicCompatibleReader( + resolved_base_url, + token=token, + model=normalized_model, + max_tokens=max_tokens, + temperature=temperature, + timeout_seconds=timeout_seconds, + ) + else: + resolved_token_env = token_env or ( + DEFAULT_DEEPSEEK_TOKEN_ENV if provider == "deepseek-openai" else DEFAULT_ANTHROPIC_TOKEN_ENV + ) + active_transport = transport + + try: + output_dir.mkdir(parents=True, exist_ok=False) + except FileExistsError as error: + raise ReaderSmokeError(f"Refusing to overwrite Reader artifacts: {output_dir}") from error + except OSError as error: + raise ReaderSmokeError(f"Cannot create Reader artifact directory: {output_dir}") from error + + manifest_path = output_dir / "reader-manifest.json" + outputs_path = output_dir / "reader-outputs.jsonl" + failures_path = output_dir / "reader-failures.jsonl" + summary_path = output_dir / "reader-summary.json" + _write_json_exclusive( + manifest_path, + { + "schema": READER_MANIFEST_SCHEMA, + "classification": "smoke-subset-reader-only", + "prepared": { + "directory": str(prepared_dir.resolve()), + "manifest_sha256": _file_digest(prepared_dir / "prepare-manifest.json"), + "prompts_sha256": _file_digest(prompts_path), + "summary_sha256": _file_digest(prepared_dir / "prepare-summary.json"), + }, + "reader": { + "provider": provider, + "base_url_env": base_url_env, + "base_url": None if base_url is None else "configured-directly", + "token_env": resolved_token_env, + "model": normalized_model, + "max_tokens": max_tokens, + "temperature": temperature, + "timeout_seconds": timeout_seconds, + }, + "judge": None, + "telemetry": "disabled-by-reader-runner", + "cost_policy": cost_policy_record(price_policy), + }, + ) + + started_ns = time.perf_counter_ns() + succeeded = 0 + failed = 0 + account = UsageAccount() + _create_empty(outputs_path) + _create_empty(failures_path) + for prompt in _jsonl(prompts_path): + if max_questions is not None and succeeded + failed >= max_questions: + break + question_id = _nonblank(prompt.get("question_id"), "question_id") + try: + output = _run_one(active_transport, prompt) + except ReaderResponseError as error: + # The transport completed and reported usage; keep the failed call priced + # and record its evidence instead of silently reporting zero cost. + failed += 1 + account.account(error.usage) + failure: dict[str, object] = { + "schema": READER_FAILURE_SCHEMA, + "question_id": question_id, + "phase": "reader", + "error_type": type(error).__name__, + "summary": (str(error).strip() or type(error).__name__)[:500], + "usage": error.usage, + } + if error.latency_ms is not None: + failure["reader_latency_ms"] = error.latency_ms + _append_json(failures_path, failure) + continue + except Exception as error: # noqa: BLE001 - each request needs an independent classified failure + failed += 1 + _append_json( + failures_path, + { + "schema": READER_FAILURE_SCHEMA, + "question_id": question_id, + "phase": "reader", + "error_type": type(error).__name__, + "summary": (str(error).strip() or type(error).__name__)[:500], + }, + ) + continue + _append_json(outputs_path, output) + succeeded += 1 + account.account(output["usage"]) + _write_json_exclusive( + summary_path, + { + "schema": READER_SUMMARY_SCHEMA, + "classification": "smoke-subset-reader-only", + "completed_at": datetime.now(UTC).isoformat(), + "question_count": succeeded + failed, + "succeeded": succeeded, + "failed": failed, + "usage": _reader_usage_totals( + input_tokens=account.input_tokens, + output_tokens=account.output_tokens, + cache_hit_tokens=account.cache_hit_tokens, + cache_miss_tokens=account.cache_miss_tokens, + cache_split_reported=account.cache_split_reported, + ), + "cost": usage_cost_block( + price_policy, + provider=provider, + model=normalized_model, + input_tokens=account.input_tokens, + cache_hit_tokens=account.cache_hit_tokens if account.cache_split_reported else None, + cache_miss_tokens=account.cache_miss_tokens if account.cache_split_reported else None, + output_tokens=account.output_tokens, + ), + "judge": None, + "elapsed_ms": round((time.perf_counter_ns() - started_ns) / 1_000_000, 3), + }, + ) + if failed: + raise ReaderSmokeError(f"Reader failed for {failed} prepared prompt(s)") + return ReaderSmokeRun( + output_dir=output_dir, + manifest_path=manifest_path, + outputs_path=outputs_path, + failures_path=failures_path, + summary_path=summary_path, + ) + + +def _run_one(transport: ReaderTransport, prompt: Mapping[str, object]) -> dict[str, object]: + question_id = _nonblank(prompt.get("question_id"), "question_id") + messages = prompt.get("messages") + if not isinstance(messages, list) or len(messages) != 2: + raise ReaderSmokeError("prepared prompt must contain exactly system and user messages") + system = messages[0] + user = messages[1] + if not isinstance(system, Mapping) or system.get("role") != "system" or not isinstance(system.get("content"), str): + raise ReaderSmokeError("prepared system message is invalid") + if not isinstance(user, Mapping) or user.get("role") != "user" or not isinstance(user.get("content"), list): + raise ReaderSmokeError("prepared user message is invalid") + raw_content = user.get("content") + if not isinstance(raw_content, list): + raise ReaderSmokeError("prepared user message is invalid") + content: list[dict[str, object]] = [] + for item in raw_content: + if not isinstance(item, Mapping): + raise ReaderSmokeError("prepared user content item must be an object") + item_type = item.get("type") + if item_type == "text": + text = item.get("text") + if not isinstance(text, str): + raise ReaderSmokeError("prepared text content is invalid") + content.append({"type": "text", "text": text}) + continue + if item_type == "image_url": + image_url = item.get("image_url") + image_value = image_url.get("url") if isinstance(image_url, Mapping) else None + if not isinstance(image_value, str): + raise ReaderSmokeError("prepared image content is invalid") + content.append({"type": "image_url", "image_url": {"url": image_value}}) + continue + raise ReaderSmokeError("prepared user content type is unsupported") + system_content = system.get("content") + if not isinstance(system_content, str): + raise ReaderSmokeError("prepared system message is invalid") + started_ns = time.perf_counter_ns() + try: + response = transport.complete(system=system_content, content=content) + except ReaderResponseError as error: + error.latency_ms = round((time.perf_counter_ns() - started_ns) / 1_000_000, 3) + raise + latency_ms = round((time.perf_counter_ns() - started_ns) / 1_000_000, 3) + raw_usage = response.get("usage") + usage = _reader_usage(raw_usage if isinstance(raw_usage, Mapping) else {}) + try: + answer = _response_text(response) + except ReaderSmokeError as error: + raise ReaderResponseError(str(error), usage=usage, latency_ms=latency_ms) from error + return { + "schema": READER_OUTPUT_SCHEMA, + "question_id": question_id, + "sequence": prompt.get("sequence"), + "domain": prompt.get("domain"), + "scope_id": prompt.get("scope_id"), + "haystack_digest": prompt.get("haystack_digest"), + "prompt_sha256": prompt.get("prompt_sha256"), + "response_text": answer, + "response_sha256": hashlib.sha256(answer.encode()).hexdigest(), + "usage": usage, + "model": response.get("model"), + "stop_reason": response.get("stop_reason"), + "reader_latency_ms": latency_ms, + } + + +def _anthropic_content(content: list[dict[str, object]]) -> list[dict[str, object]]: + converted: list[dict[str, object]] = [] + for item in content: + if item.get("type") != "image_url": + converted.append(item) + continue + image_url = item.get("image_url") + value = image_url.get("url") if isinstance(image_url, Mapping) else None + if not isinstance(value, str) or not value.startswith("data:") or ";base64," not in value: + raise ReaderSmokeError("Anthropic-compatible Reader requires base64 data URLs for image content") + header, encoded = value.split(",", 1) + media_type = header[5 : -len(";base64")] + if not media_type or not encoded: + raise ReaderSmokeError("prepared image data URL is invalid") + converted.append( + { + "type": "image", + "source": {"type": "base64", "media_type": media_type, "data": encoded}, + } + ) + return converted + + +def _reader_usage(usage: Mapping[Any, Any]) -> dict[str, object]: + """Keep a transport's cache hit/miss split so the recorded cost can price it.""" + + record: dict[str, object] = { + "input_tokens": _nonnegative_int(usage.get("input_tokens")), + "output_tokens": _nonnegative_int(usage.get("output_tokens")), + } + hit = _optional_nonnegative_int(usage.get("input_cache_hit_tokens")) + miss = _optional_nonnegative_int(usage.get("input_cache_miss_tokens")) + if hit is not None and miss is not None: + record["input_cache_hit_tokens"] = hit + record["input_cache_miss_tokens"] = miss + return record + + +def _reader_usage_totals( + *, + input_tokens: int, + output_tokens: int, + cache_hit_tokens: int, + cache_miss_tokens: int, + cache_split_reported: bool, +) -> dict[str, object]: + """Report summed usage, keeping the cache split only when every response reported it.""" + + totals: dict[str, object] = {"input_tokens": input_tokens, "output_tokens": output_tokens} + if cache_split_reported: + totals["input_cache_hit_tokens"] = cache_hit_tokens + totals["input_cache_miss_tokens"] = cache_miss_tokens + return totals + + +def _response_text(response: Mapping[str, object]) -> str: + content = response.get("content") + if not isinstance(content, list): + raise ReaderSmokeError("Reader response content must be an array") + parts = [item.get("text") for item in content if isinstance(item, Mapping) and item.get("type") == "text"] + text = "".join(part for part in parts if isinstance(part, str)).strip() + if not text: + raise ReaderSmokeError("Reader response contains no text") + return text + + +def _normalize_deepseek_response(value: object) -> Mapping[str, object]: + if not isinstance(value, Mapping): + raise ReaderSmokeError("Reader returned a non-object response") + # The HTTP call already completed, so its usage stands even when the response + # body cannot produce text; extract it before any content validation raises. + raw_usage = value.get("usage") + usage = _deepseek_usage(raw_usage if isinstance(raw_usage, Mapping) else {}) + choices = value.get("choices") + if not isinstance(choices, list) or not choices or not isinstance(choices[0], Mapping): + raise ReaderResponseError("Reader response choices are invalid", usage=usage) + choice = choices[0] + message = choice.get("message") + if not isinstance(message, Mapping): + raise ReaderResponseError("Reader response message is invalid", usage=usage) + text = message.get("content") + if not isinstance(text, str) or not text.strip(): + raise ReaderResponseError("Reader response contains no text", usage=usage) + return { + "model": value.get("model"), + "stop_reason": choice.get("finish_reason"), + "content": [{"type": "text", "text": text}], + "usage": usage, + } + + +def _deepseek_usage(usage: Mapping[Any, Any]) -> dict[str, object]: + """Keep DeepSeek's native cache split instead of collapsing it into one input total. + + ``prompt_cache_hit_tokens`` and ``prompt_cache_miss_tokens`` are billed at different + rates, so both are preserved; the total input is their sum, matching the API contract. + """ + + hit = _optional_nonnegative_int(usage.get("prompt_cache_hit_tokens")) + miss = _optional_nonnegative_int(usage.get("prompt_cache_miss_tokens")) + total = _nonnegative_int(usage.get("prompt_tokens")) + if hit is None or miss is None: + return {"input_tokens": total, "output_tokens": _nonnegative_int(usage.get("completion_tokens"))} + return { + "input_tokens": total, + "input_cache_hit_tokens": hit, + "input_cache_miss_tokens": miss, + "output_tokens": _nonnegative_int(usage.get("completion_tokens")), + } + + +def _validate_prepared_artifacts(manifest: dict[str, object], summary: dict[str, object]) -> None: + if manifest.get("classification") != "smoke-subset-prepare-only": + raise ReaderSmokeError("Prepare manifest is not a prepare-only smoke subset") + if manifest.get("reader") is not None or manifest.get("judge") is not None: + raise ReaderSmokeError("Prepared artifacts must not contain Reader or Judge execution") + if summary.get("classification") != "smoke-subset-prepare-only": + raise ReaderSmokeError("Prepare summary is not a prepare-only smoke subset") + if summary.get("question_count") != 10 or summary.get("failed") != 0: + raise ReaderSmokeError("Prepared artifacts must contain ten successful smoke prompts") + + +def _load_json(path: Path, label: str) -> dict[str, object]: + try: + value = json.loads(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error: + raise ReaderSmokeError(f"Cannot read {label}: {path}") from error + if not isinstance(value, dict): + raise ReaderSmokeError(f"{label} must be a JSON object") + return value + + +def _jsonl(path: Path) -> Iterator[dict[str, object]]: + try: + with path.open(encoding="utf-8") as stream: + for line_number, line in enumerate(stream, start=1): + value = json.loads(line) + if not isinstance(value, dict): + raise ReaderSmokeError(f"Prepared prompt {line_number} must be an object") + yield value + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error: + raise ReaderSmokeError(f"Cannot read prepared prompts: {path}") from error + + +def _file_digest(path: Path) -> str: + hasher = hashlib.sha256() + try: + with path.open("rb") as stream: + while chunk := stream.read(1024 * 1024): + hasher.update(chunk) + except OSError as error: + raise ReaderSmokeError(f"Cannot hash Reader input: {path}") from error + return hasher.hexdigest() + + +def _write_json_exclusive(path: Path, value: object) -> None: + _write_text_exclusive(path, json.dumps(value, ensure_ascii=True, indent=2, sort_keys=True) + "\n") + + +def _write_text_exclusive(path: Path, value: str) -> None: + try: + with path.open("x", encoding="utf-8", newline="") as stream: + stream.write(value) + except OSError as error: + raise ReaderSmokeError(f"Cannot write Reader artifact: {path}") from error + + +def _create_empty(path: Path) -> None: + _write_text_exclusive(path, "") + + +def _append_json(path: Path, value: object) -> None: + with path.open("a", encoding="utf-8", newline="") as stream: + stream.write(json.dumps(value, ensure_ascii=False, separators=(",", ":"), allow_nan=False) + "\n") + + +def _nonblank(value: object, label: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ReaderSmokeError(f"{label} must be a non-empty string") + return value.strip() + + +def _nonnegative_int(value: object) -> int: + return value if isinstance(value, int) and not isinstance(value, bool) and value >= 0 else 0 + + +def _optional_nonnegative_int(value: object) -> int | None: + """Keep an absent cache field distinguishable from a reported zero.""" + + if isinstance(value, int) and not isinstance(value, bool) and value >= 0: + return value + return None diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/replay_score.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/replay_score.py new file mode 100644 index 0000000000..ed5bbfe121 --- /dev/null +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/replay_score.py @@ -0,0 +1,248 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Replay deterministic LongMemEval-V2 scoring from saved Reader and Judge evidence.""" + +from __future__ import annotations + +import hashlib +import importlib +import json +import sys +import time +from collections.abc import Iterator, Mapping +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path +from typing import Any, Protocol, cast + +from powercontext_eval.benchmarks.longmemeval_v2.catalog import validate_harness_checkout +from powercontext_eval.benchmarks.longmemeval_v2.score_smoke import LLM_JUDGE_NAMES, PER_QUESTION_SCHEMA +from powercontext_eval.errors import PowerContextEvalError + +REPLAY_MANIFEST_SCHEMA = "powercontext.longmemeval-v2-score-replay.v1" +REPLAY_FAILURE_SCHEMA = "powercontext.longmemeval-v2-score-replay-failure.v1" +REPLAY_SUMMARY_SCHEMA = "powercontext.longmemeval-v2-score-replay-summary.v1" + + +class ReplayScoreError(PowerContextEvalError): + """Saved scoring evidence cannot reproduce one deterministic score result.""" + + +class MetricsAPI(Protocol): + def eval_name(self, eval_spec: str) -> str: ... + + def eval_from_spec(self, spec: str, *args: Any, **kwargs: Any) -> Any: ... + + def score_to_bool(self, value: Any) -> bool: ... + + +@dataclass(frozen=True) +class ReplayScoreRun: + output_dir: Path + manifest_path: Path + results_path: Path + failures_path: Path + summary_path: Path + + +def replay_score_smoke(*, score_dir: Path, harness_root: Path, output_dir: Path) -> ReplayScoreRun: + """Replay all score decisions from local artifacts without a Reader or Judge call.""" + + if output_dir.exists(): + raise ReplayScoreError(f"Refusing to overwrite score replay artifacts: {output_dir}") + validate_harness_checkout(harness_root) + manifest = _load_json(score_dir / "score-manifest.json", "score manifest") + summary = _load_json(score_dir / "score-summary.json", "score summary") + inputs_path = score_dir / "scoring-inputs.local.jsonl" + judge_path = score_dir / "judge-outputs.jsonl" + if not inputs_path.is_file() or not judge_path.is_file(): + raise ReplayScoreError("Score replay requires local scoring inputs and saved Judge outputs") + if manifest.get("classification") != "smoke-subset-score-only" or summary.get("failed") != 0: + raise ReplayScoreError("Score artifacts are not a successful smoke score run") + metrics = _load_metrics(harness_root) + judges = {row["question_id"]: row for row in _jsonl(judge_path, "judge output")} + try: + output_dir.mkdir(parents=True, exist_ok=False) + except FileExistsError as error: + raise ReplayScoreError(f"Refusing to overwrite score replay artifacts: {output_dir}") from error + except OSError as error: + raise ReplayScoreError(f"Cannot create score replay artifact directory: {output_dir}") from error + + manifest_path = output_dir / "replay-manifest.json" + results_path = output_dir / "replay-per-question.jsonl" + failures_path = output_dir / "replay-failures.jsonl" + summary_path = output_dir / "replay-summary.json" + _write_json_exclusive( + manifest_path, + { + "schema": REPLAY_MANIFEST_SCHEMA, + "classification": "smoke-subset-score-replay", + "source_score": { + "directory": str(score_dir.resolve()), + "manifest_sha256": _file_digest(score_dir / "score-manifest.json"), + "inputs_sha256": _file_digest(inputs_path), + "judge_outputs_sha256": _file_digest(judge_path), + "summary_sha256": _file_digest(score_dir / "score-summary.json"), + }, + }, + ) + _create_empty(results_path) + _create_empty(failures_path) + started_ns = time.perf_counter_ns() + correct = 0 + failed = 0 + total = 0 + for sequence, score_input in enumerate(_jsonl(inputs_path, "scoring input"), start=1): + total += 1 + question_id = _nonblank(score_input.get("question_id"), "question_id") + try: + result = _replay_one(metrics, score_input, judges.get(question_id), sequence) + except Exception as error: # noqa: BLE001 - preserve independent replay failure evidence + failed += 1 + _append_json( + failures_path, + { + "schema": REPLAY_FAILURE_SCHEMA, + "question_id": question_id, + "phase": "replay_score", + "error_type": type(error).__name__, + "summary": (str(error).strip() or type(error).__name__)[:500], + }, + ) + continue + _append_json(results_path, result) + result_correct = result["correct"] + if not isinstance(result_correct, bool): + raise TypeError("Replay result correct field must be boolean") + correct += int(result_correct) + _write_json_exclusive( + summary_path, + { + "schema": REPLAY_SUMMARY_SCHEMA, + "classification": "smoke-subset-score-replay", + "completed_at": datetime.now(UTC).isoformat(), + "question_count": total, + "correct": correct, + "incorrect": total - correct - failed, + "failed": failed, + "accuracy": None if failed else correct / total, + "reader_calls": 0, + "judge_calls": 0, + "elapsed_ms": round((time.perf_counter_ns() - started_ns) / 1_000_000, 3), + }, + ) + if failed: + raise ReplayScoreError(f"Score replay failed for {failed} question(s)") + return ReplayScoreRun(output_dir, manifest_path, results_path, failures_path, summary_path) + + +def _replay_one( + metrics: MetricsAPI, + score_input: Mapping[str, object], + judge: Mapping[str, object] | None, + sequence: int, +) -> dict[str, object]: + question_id = _nonblank(score_input.get("question_id"), "question_id") + eval_spec = _nonblank(score_input.get("eval_function"), "eval_function") + parsed = _nonblank(score_input.get("parsed_prediction"), "parsed_prediction") + response_hash = score_input.get("reader_response_sha256") + evaluator = metrics.eval_name(eval_spec) + if evaluator in LLM_JUDGE_NAMES: + if not isinstance(judge, Mapping) or judge.get("label") not in {0, 1}: + raise ReplayScoreError(f"Missing saved Judge label for {question_id}") + correct = judge["label"] == 1 + mode = "llm_judge_replay" + else: + answer = _nonblank(score_input.get("reference_answer"), "reference_answer") + correct = metrics.score_to_bool(metrics.eval_from_spec(eval_spec, parsed, answer)) + mode = "deterministic_replay" + return { + "schema": PER_QUESTION_SCHEMA, + "sequence": sequence, + "question_id": question_id, + "eval_function": eval_spec, + "score_mode": mode, + "correct": correct, + "parsed_prediction": parsed, + "reader_response_sha256": response_hash, + "judge_output_ref": None if evaluator not in LLM_JUDGE_NAMES else question_id, + } + + +def _load_metrics(harness_root: Path) -> MetricsAPI: + root = str(harness_root.resolve()) + if root not in sys.path: + sys.path.insert(0, root) + return cast(MetricsAPI, importlib.import_module("evaluation.qa_eval_metrics")) + + +def _load_json(path: Path, label: str) -> dict[str, object]: + try: + value = json.loads(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error: + raise ReplayScoreError(f"Cannot read {label}: {path}") from error + if not isinstance(value, dict): + raise ReplayScoreError(f"{label} must be a JSON object") + return value + + +def _jsonl(path: Path, label: str) -> Iterator[dict[str, object]]: + try: + with path.open(encoding="utf-8") as stream: + for line_number, line in enumerate(stream, start=1): + value = json.loads(line) + if not isinstance(value, dict): + raise ReplayScoreError(f"{label} row {line_number} must be an object") + yield value + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error: + raise ReplayScoreError(f"Cannot read {label} input: {path}") from error + + +def _file_digest(path: Path) -> str: + hasher = hashlib.sha256() + try: + with path.open("rb") as stream: + while chunk := stream.read(1024 * 1024): + hasher.update(chunk) + except OSError as error: + raise ReplayScoreError(f"Cannot hash replay input: {path}") from error + return hasher.hexdigest() + + +def _write_json_exclusive(path: Path, value: object) -> None: + _write_text_exclusive(path, json.dumps(value, ensure_ascii=True, indent=2, sort_keys=True) + "\n") + + +def _write_text_exclusive(path: Path, value: str) -> None: + try: + with path.open("x", encoding="utf-8", newline="") as stream: + stream.write(value) + except OSError as error: + raise ReplayScoreError(f"Cannot write replay artifact: {path}") from error + + +def _create_empty(path: Path) -> None: + _write_text_exclusive(path, "") + + +def _append_json(path: Path, value: object) -> None: + with path.open("a", encoding="utf-8", newline="") as stream: + stream.write(json.dumps(value, ensure_ascii=False, separators=(",", ":"), allow_nan=False) + "\n") + + +def _nonblank(value: object, label: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ReplayScoreError(f"{label} must be a non-empty string") + return value.strip() diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/report.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/report.py new file mode 100644 index 0000000000..6013b0a502 --- /dev/null +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/report.py @@ -0,0 +1,812 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Summarize one saved LongMemEval-V2 smoke run without calling a model or server.""" + +from __future__ import annotations + +import json +import math +from collections.abc import Iterator, Mapping +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path +from typing import TypeAlias, TypeGuard, cast + +from powercontext_eval.benchmarks.longmemeval_v2.costs import model_free_cost_block +from powercontext_eval.errors import PowerContextEvalError + +REPORT_SCHEMA = "powercontext.longmemeval-v2-smoke-report.v1" +REPORT_BANNER = "LongMemEval-V2 smoke subset — not a complete benchmark result." +REPORT_BOUNDARY = ( + "This report describes one fixed smoke subset over pinned upstream trajectories. " + "It is not a complete benchmark result, not a product reliability claim, and not a substitute " + "for a full LongMemEval-V2 run. See evaluation/README.md and " + "evaluation/docs/longmemeval-v2-full-run.md for the recorded boundaries and the unexecuted full run." +) +ERROR_CLASSES = ("configuration", "infrastructure", "retrieval", "generation", "judge", "integrity") +_PHASES = ("preflight", "retrieval", "prepare", "reader", "score", "replay") +_FAILURE_CLASS_BY_PHASE_FILE = { + "prepare": "infrastructure", + "reader": "generation", + "score": "judge", + "replay": "integrity", +} + +ReportStatus: TypeAlias = str + + +class ReportError(PowerContextEvalError): + """Saved run artifacts cannot produce one unified report.""" + + +@dataclass(frozen=True) +class ReportRun: + """The written unified report artifacts.""" + + report_path: Path + markdown_path: Path + + +def build_report(*, run_dir: Path, output_dir: Path | None = None) -> ReportRun: + """Read saved stage artifacts and write ``report.json`` and ``report.md`` fail-closed.""" + + from powercontext_eval.benchmarks.longmemeval_v2.run_smoke import PHASE_DIRECTORIES + + run_root = run_dir.resolve() + if not run_root.is_dir(): + raise ReportError(f"LongMemEval-V2 smoke run directory does not exist: {run_root}") + # Rebuild the phase mapping with plain string keys: ``Mapping`` is invariant in its key + # type, so the runner's ``Mapping[Phase, str]`` cannot be passed to the str-keyed helpers. + directories = {str(phase): name for phase, name in PHASE_DIRECTORIES.items()} + report = _build(run_root, directories) + if output_dir is None: + report_path = run_root / "report.json" + markdown_path = run_root / "report.md" + _write_text_exclusive(report_path, _json(report)) + _write_text_exclusive(markdown_path, _markdown(report)) + return ReportRun(report_path=report_path, markdown_path=markdown_path) + target = output_dir.resolve() + try: + target.mkdir(parents=True, exist_ok=False) + except FileExistsError as error: + raise ReportError(f"Refusing to overwrite report artifacts: {target}") from error + except OSError as error: + raise ReportError(f"Cannot create report artifact directory: {target}") from error + report_path = target / "report.json" + markdown_path = target / "report.md" + _write_text_exclusive(report_path, _json(report)) + _write_text_exclusive(markdown_path, _markdown(report)) + return ReportRun(report_path=report_path, markdown_path=markdown_path) + + +def _build(run_root: Path, directories: Mapping[str, str]) -> dict[str, object]: + manifest = _load_json(run_root / "run-manifest.json") + summary = _load_json(run_root / "run-summary.json") + retrieval_manifest = _load_json(run_root / directories["retrieval"] / "retrieval-manifest.json") + retrieval_summary = _load_json(run_root / directories["retrieval"] / "summary.json") + prepare_summary = _load_json(run_root / directories["prepare"] / "prepare-summary.json") + reader_summary = _load_json(run_root / directories["reader"] / "reader-summary.json") + score_summary = _load_json(run_root / directories["score"] / "score-summary.json") + replay_summary = _load_json(run_root / directories["replay"] / "replay-summary.json") + retrieval_results = run_root / directories["retrieval"] / "retrieval-results.jsonl" + audit = run_root / directories["retrieval"] / "adapter-audit.jsonl" + reader_outputs = run_root / directories["reader"] / "reader-outputs.jsonl" + judge_outputs = run_root / directories["score"] / "judge-outputs.jsonl" + + context_items = _sum_list_length(retrieval_results, "memory_context") + retrieval_latency = _sum_nested_ms(retrieval_results, "timings_ms") + ingest_latency = _sum_ingest_ms(audit) + reader_latency = _sum_number(reader_outputs, "reader_latency_ms") + judge_latency, abstention = _judge_totals(judge_outputs) + artifacts = _artifacts(run_root, directories) + usage = _usage(reader_summary, score_summary, manifest) + return { + "schema": REPORT_SCHEMA, + "classification": "smoke-subset", + "status": _status(summary, score_summary, artifacts), + "run_id": _run_id(manifest, summary), + "experiment_arm": _experiment_arm(manifest, retrieval_manifest), + "generated_at": datetime.now(UTC).isoformat(), + "question_count": _question_count( + summary, score_summary, replay_summary, reader_summary, prepare_summary, retrieval_summary + ), + "accuracy": _accuracy(score_summary, replay_summary), + "latency_ms": { + "ingest": ingest_latency, + "retrieval": retrieval_latency, + "prepare": _number(prepare_summary, "elapsed_ms"), + "reader": reader_latency, + "judge": judge_latency, + }, + "context": { + "items": context_items, + "bytes": _number(retrieval_summary, "context_bytes"), + "tokens": _number(prepare_summary, "memory_context_tokens"), + "citations_available": _number(retrieval_summary, "citation_count"), + }, + "usage": usage, + "failures": _failure_counts(run_root, directories), + "abstention": abstention, + "artifacts": artifacts, + "boundary": REPORT_BOUNDARY, + } + + +def _status( + summary: dict[str, object] | None, + score_summary: dict[str, object] | None, + artifacts: Mapping[str, object], +) -> ReportStatus: + if summary is not None and summary.get("status") in {"completed", "partial", "failed"}: + return str(summary["status"]) + if score_summary is not None and score_summary.get("failed") == 0: + return "completed" + if any(artifacts.get(phase) is not None for phase in _PHASES): + return "partial" + return "failed" + + +def _run_id(manifest: dict[str, object] | None, summary: dict[str, object] | None) -> str | None: + for source in (manifest, summary): + if source is None: + continue + value = source.get("run_id") + if isinstance(value, str) and value.strip(): + return value + return None + + +def _experiment_arm( + manifest: dict[str, object] | None, retrieval_manifest: dict[str, object] | None +) -> dict[str, object] | None: + """Read the recorded arm identity, preferring the run manifest over a retrieval-only run.""" + + for source in (manifest, retrieval_manifest): + if source is None: + continue + value = source.get("experiment_arm") + if isinstance(value, dict): + return {str(key): item for key, item in value.items()} + return None + + +def _usage( + reader_summary: dict[str, object] | None, + score_summary: dict[str, object] | None, + manifest: dict[str, object] | None, +) -> dict[str, object]: + """Report tokens, per-stage usage-based cost, and the ingestion zero-cost record. + + The total is reported only when every model stage that actually ran has a priced cost + recorded under one price policy revision. A stage counts as run when its summary + exists, or when the manifest modes say it was configured and no summary proves + otherwise — a missing or unpriced stage cost therefore keeps the total ``null`` + instead of silently summing the stages that happen to be present. + """ + + reader_cost = _stage_cost(reader_summary, "cost") + judge_cost = _stage_cost(score_summary, "judge_cost") + ingestion = model_free_cost_block( + stage="ingestion", + reason=( + "the PowerContext Memory adapter ingests without a model, so no provider reports " + "ingestion usage and ingestion cost is zero" + ), + ) + modes = _mapping(manifest.get("modes")) if manifest is not None else {} + judge_calls = _judge_call_count(score_summary) + reader_state = _model_stage_state( + summary=reader_summary, + cost=reader_cost, + configured=modes.get("reader") is True, + calls_expected=False, + role="Reader", + configured_provider=_configured_provider(manifest, "reader"), + configured_model=_configured_model(manifest, "reader"), + configured_policy=_configured_cost_policy(manifest, "reader"), + ) + judge_state = _model_stage_state( + summary=score_summary, + cost=judge_cost, + configured=modes.get("score") is True, + calls_expected=judge_calls > 0, + role="Judge", + configured_provider=_configured_provider(manifest, "judge"), + configured_model=_configured_model(manifest, "judge"), + configured_policy=_configured_cost_policy(manifest, "judge"), + recorded_usage_nonzero=_mapping_nonzero(score_summary, "judge_usage"), + calls=judge_calls, + ) + total, note = _total_cost( + reader_cost, + judge_cost, + reader_state=reader_state, + judge_state=judge_state, + include_judge=judge_calls > 0, + ) + return { + "ingestion_tokens": 0, + "ingestion_tokens_note": ( + "the PowerContext Memory adapter ingests without a model, so no provider reports ingestion usage" + ), + "ingestion_cost": ingestion, + "reader_input_tokens": _nested_number(reader_summary, "usage", "input_tokens"), + "reader_output_tokens": _nested_number(reader_summary, "usage", "output_tokens"), + "judge_input_tokens": _nested_number(score_summary, "judge_usage", "input_tokens"), + "judge_output_tokens": _nested_number(score_summary, "judge_usage", "output_tokens"), + "reader_cost": reader_cost, + "judge_cost": judge_cost, + "estimated_cost_usd": None if total is None else total["cost_usd"], + "estimated_cost_note": note, + } + + +def _model_stage_state( + *, + summary: dict[str, object] | None, + cost: Mapping[str, object] | None, + configured: bool, + calls_expected: bool, + role: str, + configured_provider: str | None, + configured_model: str | None, + configured_policy: Mapping[str, object] | None, + recorded_usage_nonzero: bool = False, + calls: int = 0, +) -> str | None: + """Return the reason this model stage blocks a total, or ``None`` when it is settled. + + ``None`` also means the stage did not run and therefore owes no cost: only a stage whose + summary exists, or that the manifest modes configured without a contrary summary, must + account for its cost. + """ + + if summary is None: + if not configured: + return None + return f"the run was configured to run the {role} but no {role} summary recorded a cost" + if role == "Judge" and not calls_expected: + if recorded_usage_nonzero or (cost is not None and _nonnull_usage(cost)): + return "the Judge recorded non-zero usage despite zero recorded model calls" + return None + if cost is None: + # A score stage that made no model call owes no Judge cost; any other absent block is + # a missing artifact the report must not silently drop. + if role == "Judge" and not calls_expected: + return None + if calls_expected: + return f"the {role} made {calls} model call(s) but recorded no priced cost" + return f"the {role} stage ran but its summary recorded no cost block" + amount = _cost_amount(cost) + if amount is None: + if not calls_expected and role == "Judge" and not _nonnull_usage(cost): + # No Judge call happened and the recorded usage is zero, so the absent price is + # genuinely zero cost rather than a missing artifact. + return None + reason = cost.get("unavailable_reason") + detail = f": {reason}" if reason else "" + if calls_expected: + return f"the {role} made {calls} model call(s) but recorded no priced cost{detail}" + if _nonnull_usage(cost): + return f"the {role} recorded non-zero usage without a priced cost{detail}" + return f"the {role} stage recorded no priced cost{detail}" + identity_problem = _cost_identity_problem( + cost, + configured_provider=configured_provider, + configured_model=configured_model, + configured_policy=configured_policy, + ) + if identity_problem is not None: + return f"the {role} cost {identity_problem}" + if configured_model is not None and str(cost.get("model")) != configured_model: + return f"the {role} cost was recorded for model {cost.get('model')!r}, not the configured {configured_model!r}" + return None + + +def _nonnull_usage(cost: Mapping[str, object]) -> bool: + """Report whether a cost block carries any non-zero token count.""" + + for key in ("input_tokens", "input_cache_hit_tokens", "input_cache_miss_tokens", "output_tokens"): + value = cost.get(key) + if isinstance(value, int) and not isinstance(value, bool) and value > 0: + return True + return False + + +def _judge_call_count(score_summary: dict[str, object] | None) -> int: + """Count real Judge model calls, falling back to recorded usage for older artifacts.""" + + if score_summary is None: + return 0 + value = score_summary.get("judge_calls") + if isinstance(value, int) and not isinstance(value, bool) and value >= 0: + return value + usage = score_summary.get("judge_usage") + if isinstance(usage, Mapping) and any( + isinstance(token_count, int) and not isinstance(token_count, bool) and token_count > 0 + for token_count in usage.values() + ): + return 1 + cost = score_summary.get("judge_cost") + if isinstance(cost, Mapping): + return 1 if _cost_amount({str(name): item for name, item in cost.items()}) is not None else 0 + return 0 + + +def _configured_model(manifest: dict[str, object] | None, role: str) -> str | None: + """Read the model the run manifest pinned for one role, when it recorded one.""" + + if manifest is None: + return None + block = manifest.get(role) + if not isinstance(block, Mapping): + return None + model = block.get("model") + return model.strip() if isinstance(model, str) and model.strip() else None + + +def _configured_provider(manifest: dict[str, object] | None, role: str) -> str | None: + if manifest is None: + return None + block = manifest.get(role) + if not isinstance(block, Mapping): + return None + provider = block.get("provider") + return provider.strip() if isinstance(provider, str) and provider.strip() else None + + +def _configured_cost_policy(manifest: dict[str, object] | None, role: str) -> Mapping[str, object] | None: + if manifest is None: + return None + policies = manifest.get("cost_policy") + if not isinstance(policies, Mapping): + return None + policy = policies.get(role.lower()) + if not isinstance(policy, Mapping): + return None + return {str(key): value for key, value in policy.items()} + + +def _mapping_nonzero(summary: dict[str, object] | None, key: str) -> bool: + if summary is None: + return False + usage = summary.get(key) + if not isinstance(usage, Mapping): + return False + return any(isinstance(value, int) and not isinstance(value, bool) and value > 0 for value in usage.values()) + + +def _cost_identity_problem( + cost: Mapping[str, object], + *, + configured_provider: str | None, + configured_model: str | None, + configured_policy: Mapping[str, object] | None, +) -> str | None: + if cost.get("currency") != "USD": + return "was not recorded in USD" + if configured_provider is not None and cost.get("provider") != configured_provider: + return f"was recorded for provider {cost.get('provider')!r}, not the configured {configured_provider!r}" + if configured_model is not None and cost.get("model") != configured_model: + return f"was recorded for model {cost.get('model')!r}, not the configured {configured_model!r}" + if configured_policy is None: + return "was priced but the run manifest has no configured price policy" + for key in ( + "provider", + "model", + "currency", + "input_cache_hit_price_per_million", + "input_cache_miss_price_per_million", + "output_price_per_million", + "price_policy_revision", + ): + if cost.get(key) != configured_policy.get(key): + return f"does not match the run's configured price policy field {key}" + input_tokens = cost.get("input_tokens") + hit = cost.get("input_cache_hit_tokens") + miss = cost.get("input_cache_miss_tokens") + if not all( + isinstance(value, int) and not isinstance(value, bool) and value >= 0 for value in (input_tokens, hit, miss) + ): + return "has invalid cache token counts" + normalized_input = cast(int, input_tokens) + normalized_hit = cast(int, hit) + normalized_miss = cast(int, miss) + if normalized_input != normalized_hit + normalized_miss: + return "has cache token counts that do not equal its input token total" + return None + + +def _stage_cost(summary: dict[str, object] | None, key: str) -> dict[str, object] | None: + """Read one stage's cost block, preserving an unconfigured cost as null with its reason.""" + + if summary is None: + return None + value = summary.get(key) + if not isinstance(value, Mapping): + return None + record = {str(name): item for name, item in value.items()} + cost = record.get("cost_usd") + if cost is not None and not _finite_nonnegative_number(cost): + return None + return record + + +def _total_cost( + reader_cost: Mapping[str, object] | None, + judge_cost: Mapping[str, object] | None, + *, + reader_state: str | None, + judge_state: str | None, + include_judge: bool, +) -> tuple[dict[str, object] | None, str]: + """Sum stage costs only when every model stage that ran was priced under one policy.""" + + problems = [state for state in (reader_state, judge_state) if state] + if problems: + return None, "; ".join(problems) + costs = [ + cost + for cost in (reader_cost, judge_cost if include_judge else None) + if cost is not None and _cost_amount(cost) is not None + ] + if not costs: + return None, "no Reader or Judge model stage ran in this smoke run, so no model cost is reported" + currencies = {str(cost.get("currency")) for cost in costs} + if len(currencies) != 1: + return None, "the recorded stage costs use different currencies, so no single total is reported" + revisions = {str(cost.get("price_policy_revision")) for cost in costs} + if len(revisions) != 1: + return None, "the recorded stage costs were priced under different price policy revisions" + amounts = [_cost_amount(cost) for cost in costs] + return ( + {"cost_usd": round(sum(amount for amount in amounts if amount is not None), 6), "currency": currencies.pop()}, + "sum of the Reader and Judge usage costs recorded under the run's explicit price policy", + ) + + +def _cost_amount(cost: Mapping[str, object]) -> float | None: + """Read one recorded cost amount, keeping a null cost distinguishable from zero.""" + + value = cost.get("cost_usd") + if not _finite_nonnegative_number(value): + return None + return float(value) + + +def _finite_nonnegative_number(value: object) -> TypeGuard[int | float]: + return ( + isinstance(value, (int, float)) and not isinstance(value, bool) and math.isfinite(float(value)) and value >= 0 + ) + + +def _question_count(*summaries: dict[str, object] | None) -> int | None: + for summary in summaries: + if summary is None: + continue + value = summary.get("question_count") + if _is_int(value): + return value + return None + + +def _accuracy( + score_summary: dict[str, object] | None, replay_summary: dict[str, object] | None +) -> dict[str, object] | None: + for summary in (score_summary, replay_summary): + if summary is None or summary.get("failed") != 0: + continue + correct = summary.get("correct") + incorrect = summary.get("incorrect") + if not _is_int(correct) or not _is_int(incorrect): + continue + value = summary.get("accuracy") + return { + "correct": correct, + "incorrect": incorrect, + "failed": 0, + "value": float(value) if isinstance(value, (int, float)) and not isinstance(value, bool) else None, + } + return None + + +def _failure_counts(run_root: Path, directories: Mapping[str, str]) -> dict[str, int]: + counts = dict.fromkeys(ERROR_CLASSES, 0) + for row in _iter_jsonl(run_root / "failures.jsonl"): + error_class = row.get("error_class") + if error_class in counts: + counts[str(error_class)] += 1 + for phase, error_class in _FAILURE_CLASS_BY_PHASE_FILE.items(): + for row in _iter_jsonl(run_root / directories[phase] / f"{phase}-failures.jsonl"): + counts[_phase_failure_class(phase, row, error_class)] += 1 + for row in _iter_jsonl(run_root / directories["retrieval"] / "failures.jsonl"): + counts[_phase_failure_class("retrieval", row, "infrastructure")] += 1 + return counts + + +def _phase_failure_class(phase: str, row: Mapping[str, object], fallback: str) -> str: + if phase != "retrieval": + return fallback + category = row.get("category") + if category == "integration": + return "retrieval" + if category == "integrity": + return "integrity" + return "infrastructure" + + +def _judge_totals(path: Path) -> tuple[float | None, dict[str, object]]: + latency = 0.0 + seen = False + count = 0 + correct = 0 + available = False + for row in _iter_jsonl(path): + seen = True + value = row.get("judge_latency_ms") + if isinstance(value, (int, float)) and not isinstance(value, bool): + latency += float(value) + if row.get("evaluator") != "llm_abstention_checker": + continue + available = True + count += 1 + if row.get("label") == 1: + correct += 1 + abstention: dict[str, object] = { + "count": count, + "correct": correct if available else None, + "incorrect": count - correct if available else None, + "unavailable": not available, + } + return (round(latency, 3) if seen else None), abstention + + +def _artifacts(run_root: Path, directories: Mapping[str, str]) -> dict[str, str | None]: + candidates: dict[str, str] = { + "run_manifest": "run-manifest.json", + "run_summary": "run-summary.json", + "failures": "failures.jsonl", + } + candidates.update(dict(directories)) + return {name: value if (run_root / value).exists() else None for name, value in candidates.items()} + + +def _number(source: dict[str, object] | None, key: str) -> int | float | None: + """Preserve a saved integer count as an integer and never invent a value that was not recorded.""" + + if source is None: + return None + value = source.get(key) + if isinstance(value, bool): + return None + if isinstance(value, int): + return value + if isinstance(value, float): + return value + return None + + +def _nested_number(source: dict[str, object] | None, key: str, nested: str) -> int | None: + if source is None: + return None + inner = source.get(key) + if not isinstance(inner, Mapping): + return None + value = inner.get(nested) + return value if _is_int(value) else None + + +def _sum_list_length(path: Path, key: str) -> int | None: + total = 0 + seen = False + for row in _iter_jsonl(path): + seen = True + value = row.get(key) + if isinstance(value, list): + total += len(value) + return total if seen else None + + +def _sum_number(path: Path, key: str) -> float | None: + total = 0.0 + seen = False + for row in _iter_jsonl(path): + seen = True + value = row.get(key) + if isinstance(value, (int, float)) and not isinstance(value, bool): + total += float(value) + return round(total, 3) if seen else None + + +def _sum_nested_ms(path: Path, key: str) -> float | None: + total = 0.0 + seen = False + for row in _iter_jsonl(path): + inner = row.get(key) + if not isinstance(inner, Mapping): + continue + value = inner.get("total") + if isinstance(value, (int, float)) and not isinstance(value, bool): + seen = True + total += float(value) + return round(total, 3) if seen else None + + +def _sum_ingest_ms(path: Path) -> float | None: + total = 0.0 + seen = False + for row in _iter_jsonl(path): + if row.get("operation") != "ingest": + continue + inner = row.get("timings_ms") + if not isinstance(inner, Mapping): + continue + value = inner.get("total") + if isinstance(value, (int, float)) and not isinstance(value, bool): + seen = True + total += float(value) + return round(total, 3) if seen else None + + +def _is_int(value: object) -> TypeGuard[int]: + return isinstance(value, int) and not isinstance(value, bool) + + +def _load_json(path: Path) -> dict[str, object] | None: + try: + value = json.loads(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError): + return None + return value if isinstance(value, dict) else None + + +def _iter_jsonl(path: Path) -> Iterator[dict[str, object]]: + try: + with path.open(encoding="utf-8") as stream: + for line in stream: + if not line.strip(): + continue + try: + value = json.loads(line) + except (UnicodeDecodeError, json.JSONDecodeError): + continue + if isinstance(value, dict): + yield value + except OSError: + return + + +def _json(value: object) -> str: + return json.dumps(value, ensure_ascii=True, indent=2, sort_keys=True, allow_nan=False) + "\n" + + +def _markdown(report: Mapping[str, object]) -> str: + accuracy = _mapping(report["accuracy"]) + latency = _mapping(report["latency_ms"]) + context = _mapping(report["context"]) + usage = _mapping(report["usage"]) + failures = _mapping(report["failures"]) + abstention = _mapping(report["abstention"]) + artifacts = _mapping(report["artifacts"]) + lines = [ + REPORT_BANNER, + "", + f"Status: {report['status']}", + f"Run: {_display(report['run_id'])}", + f"Arm: {_arm_id(report['experiment_arm'])}", + f"Questions: {_display(report['question_count'])}", + f"Accuracy: {_display_accuracy(accuracy)}", + "", + "Latency (ms): " + _pairs(latency), + "Context: " + _pairs(context), + "Usage: " + + _pairs( + {name: value for name, value in usage.items() if name.endswith("tokens") or name == "estimated_cost_usd"} + ), + "Cost: " + _pairs(_cost_line(usage)), + "Failures: " + _pairs(failures), + "Abstention: " + _pairs(abstention), + "", + "Artifacts:", + ] + lines.extend(f"- {name}: {_display(value)}" for name, value in sorted(artifacts.items())) + lines.extend( + [ + "", + "What this evaluates:", + "- Whether the PowerContext Memory adapter retrieved citable evidence through public interfaces", + " under fixed upstream data, fixed questions, and a fixed context budget.", + "- Answer accuracy, latency, context size, failures, and abstention for the configured Reader and Judge.", + "- Whether saved outputs replay the same deterministic scoring inputs without a model.", + "", + "What this does not evaluate:", + "- Handoff, cross-host recovery, normal Runtime persistence, or Work Continuity.", + "- Full LongMemEval-V2 performance, general model capability, or product leadership.", + "- LoCoMo, SWE-bench Pro, or real user-task acceptance.", + "", + REPORT_BOUNDARY, + "", + ] + ) + return "\n".join(lines) + + +def _mapping(value: object) -> Mapping[str, object]: + """Rebuild an arbitrary mapping with string keys so downstream lookups stay typed.""" + + if not isinstance(value, Mapping): + return {} + return {str(key): item for key, item in value.items()} + + +def _arm_id(value: object) -> str: + if isinstance(value, Mapping): + identifier = value.get("id") + if isinstance(identifier, str) and identifier.strip(): + return identifier + return "unavailable" + + +def _cost_line(usage: Mapping[str, object]) -> dict[str, object]: + """Render one readable cost line without leaking whole nested stage records.""" + + reader = usage.get("reader_cost") + judge = usage.get("judge_cost") + return { + "ingestion": _stage_cost_display(usage.get("ingestion_cost")), + "reader": _stage_cost_display(reader), + "judge": _stage_cost_display(judge), + "total_usd": usage.get("estimated_cost_usd"), + } + + +def _stage_cost_display(value: object) -> object: + if not isinstance(value, Mapping): + return None + cost = value.get("cost_usd") + if cost is None: + reason = value.get("unavailable_reason") + return f"unavailable ({reason})" if reason else "unavailable" + currency = value.get("currency") + return str(cost) if currency is None else f"{cost} {currency}" + + +def _pairs(values: Mapping[str, object]) -> str: + return ", ".join(f"{name}={_display(value)}" for name, value in sorted(values.items())) + + +def _display(value: object) -> str: + return "unavailable" if value is None else str(value) + + +def _display_accuracy(accuracy: Mapping[str, object]) -> str: + value = accuracy.get("value") + correct = accuracy.get("correct") + incorrect = accuracy.get("incorrect") + if value is None: + return "unavailable" + if isinstance(correct, bool) or not isinstance(correct, int): + return "unavailable" + if isinstance(incorrect, bool) or not isinstance(incorrect, int): + return "unavailable" + return f"{correct}/{correct + incorrect} ({value})" + + +def _write_text_exclusive(path: Path, value: str) -> None: + try: + with path.open("x", encoding="utf-8", newline="") as stream: + stream.write(value) + except OSError as error: + raise ReportError(f"Cannot write report artifact: {path}") from error diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/retrieval_smoke.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/retrieval_smoke.py new file mode 100644 index 0000000000..9b9d356a32 --- /dev/null +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/retrieval_smoke.py @@ -0,0 +1,678 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Run the fixed LongMemEval-V2 subset through PowerContext without a Reader or Judge.""" + +from __future__ import annotations + +import hashlib +import json +import os +import time +import uuid +from collections.abc import Iterator, Mapping +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path +from typing import Literal, Protocol, cast + +from powercontext_eval.benchmarks.longmemeval_v2.adapter import ( + DEFAULT_SOURCE_CHUNK_BYTES, + MAX_MEMORY_TEXT_BYTES, + PowerContextHTTPRuntime, + PowerContextMemory, + PowerContextMemoryAdapterError, + PowerContextMemoryModeError, + PowerContextRuntime, +) +from powercontext_eval.benchmarks.longmemeval_v2.arms import ( + ExperimentArm, + arm_manifest_record, + resolve_experiment_arm, +) +from powercontext_eval.benchmarks.longmemeval_v2.catalog import ( + Ability, + SmokeSelection, + load_dataset_lock, + load_smoke_manifest, +) +from powercontext_eval.benchmarks.longmemeval_v2.smoke import prepare_smoke_run +from powercontext_eval.errors import PowerContextEvalError + +RETRIEVAL_MANIFEST_SCHEMA = "powercontext.longmemeval-v2-retrieval-run.v1" +RETRIEVAL_RESULT_SCHEMA = "powercontext.longmemeval-v2-retrieval-result.v1" +RETRIEVAL_FAILURE_SCHEMA = "powercontext.longmemeval-v2-retrieval-failure.v1" +RETRIEVAL_SUMMARY_SCHEMA = "powercontext.longmemeval-v2-retrieval-summary.v1" + + +class RetrievalSmokeError(PowerContextEvalError): + """A retrieval-only smoke run could not preserve its evaluation contract.""" + + +class RetrievalCapabilityError(RetrievalSmokeError): + """The Server cannot execute the search mode required by the experiment arm.""" + + +class RetrievalSmokeRuntime(PowerContextRuntime, Protocol): + """Public PowerContext operations required by the retrieval runner.""" + + def create_scope(self, payload: Mapping[str, object]) -> Mapping[str, object]: ... + + def get_readiness(self) -> Mapping[str, object]: ... + + def get_capabilities(self) -> Mapping[str, object]: ... + + +@dataclass(frozen=True) +class RetrievalQuestion: + question_id: str + domain: Literal["web", "enterprise"] + ability: Ability + text: str + image: str | None + + +@dataclass(frozen=True) +class HaystackGroup: + digest: str + trajectory_ids: tuple[str, ...] + question_ids: tuple[str, ...] + + +@dataclass(frozen=True) +class RetrievalSmokeRun: + output_dir: Path + manifest_path: Path + results_path: Path + failures_path: Path + summary_path: Path + audit_path: Path + + +def run_retrieval_smoke( + *, + data_root: Path, + dataset_lock: Path, + harness_root: Path, + smoke_manifest: Path, + output_dir: Path, + run_id: str, + powercontext_revision: str, + integration_revision: str, + base_url: str = "http://127.0.0.1:8000", + token_env: str = "POWERCONTEXT_TOKEN", + experiment_arm: str | ExperimentArm | None = None, + search_limit: int = 10, + timeout_seconds: float = 30.0, + runtime: RetrievalSmokeRuntime | None = None, +) -> RetrievalSmokeRun: + """Ingest the fixed subset by isolated haystack and record retrieval-only outputs.""" + + if output_dir.exists(): + raise RetrievalSmokeError(f"Refusing to overwrite retrieval smoke artifacts: {output_dir}") + normalized_run_id = _nonblank(run_id, "run_id") + powercontext_ref = _nonblank(powercontext_revision, "powercontext_revision") + integration_ref = _nonblank(integration_revision, "integration_revision") + arm = resolve_experiment_arm(experiment_arm) + if isinstance(search_limit, bool) or not 1 <= search_limit <= 50: + raise RetrievalSmokeError("search_limit must be from 1 through 50") + if timeout_seconds <= 0: + raise RetrievalSmokeError("timeout_seconds must be positive") + + client: RetrievalSmokeRuntime = runtime or PowerContextHTTPRuntime( + base_url, + token=os.getenv(token_env), + timeout_seconds=timeout_seconds, + ) + _require_runtime(client, arm) + started_at = datetime.now(UTC) + started_ns = time.perf_counter_ns() + # Per-execution scope namespace: a run_id alone collides across repeated runs, + # experiment arms, and same-basename output directories, silently sharing Scopes. + execution_namespace = uuid.uuid4().hex + prepared = prepare_smoke_run( + data_root=data_root, + dataset_lock=dataset_lock, + harness_root=harness_root, + smoke_manifest=smoke_manifest, + output_dir=output_dir, + ) + lock = load_dataset_lock(dataset_lock) + selection = load_smoke_manifest(smoke_manifest) + questions = _load_questions( + data_root / "questions.jsonl", + selection, + expected_digest=lock.file_digests["questions.jsonl"], + ) + groups, question_groups = _load_groups( + data_root / "haystacks" / f"lme_v2_{selection.tier}.json", + selection, + expected_digest=lock.file_digests[f"haystacks/lme_v2_{selection.tier}.json"], + ) + + results_path = output_dir / "retrieval-results.jsonl" + failures_path = output_dir / "failures.jsonl" + retrieval_manifest_path = output_dir / "retrieval-manifest.json" + summary_path = output_dir / "summary.json" + audit_path = output_dir / "adapter-audit.jsonl" + _create_empty(results_path) + _create_empty(failures_path) + + root_scope = _create_scope( + client, + title=f"LongMemEval-V2 retrieval smoke {normalized_run_id} ({arm.arm_id})", + summary="Retrieval-only fixed smoke subset; no Reader, Judge, or scoring.", + idempotency_seed=f"{normalized_run_id}:{arm.arm_id}:{execution_namespace}:root", + parent_scope_id=None, + ) + group_scopes: dict[str, str] = {} + adapters: dict[str, PowerContextMemory] = {} + for sequence, group in enumerate(groups, start=1): + scope_id = _create_scope( + client, + title=f"LongMemEval-V2 haystack {sequence}: {group.digest[:12]}", + summary=f"Isolated haystack with {len(group.trajectory_ids)} trajectories.", + idempotency_seed=f"{normalized_run_id}:{arm.arm_id}:{execution_namespace}:haystack:{group.digest}", + parent_scope_id=root_scope, + ) + group_scopes[group.digest] = scope_id + adapter = PowerContextMemory( + { + "scope_id": scope_id, + "audit_path": str(audit_path), + "base_url": base_url, + "token_env": token_env, + "search_mode": arm.search_mode, + "query_strategy": arm.retrieval_strategy, + "prepared_context_max_bytes": arm.prepared_context_max_bytes or 8_000, + "memory_projection": arm.memory_projection, + "task_lens": arm.task_lens, + "search_limit": search_limit, + "timeout_seconds": timeout_seconds, + } + ) + adapter.configure_runtime(runtime=client) + adapters[group.digest] = adapter + + _write_json_exclusive( + retrieval_manifest_path, + { + "schema": RETRIEVAL_MANIFEST_SCHEMA, + "classification": "smoke-subset-retrieval-only", + "run_id": normalized_run_id, + "execution_namespace": execution_namespace, + "started_at": started_at.isoformat(), + "experiment_arm": arm_manifest_record(arm), + "revisions": { + "powercontext": powercontext_ref, + "integration": integration_ref, + }, + "runtime": { + "base_url": base_url, + "token_env": token_env, + "search_mode": arm.search_mode, + "query_strategy": arm.retrieval_strategy, + "prepared_context_max_bytes": arm.prepared_context_max_bytes, + "search_limit": search_limit, + "timeout_seconds": timeout_seconds, + "reader": None, + "judge": None, + "source_projection": "full-allowed-trajectory-fields", + "source_chunk_bytes": DEFAULT_SOURCE_CHUNK_BYTES, + "memory_projection": arm.memory_projection, + "query_projection": arm.query_projection, + "memory_max_bytes": MAX_MEMORY_TEXT_BYTES, + }, + "input_artifacts": { + "manifest": prepared.manifest_path.name, + "subset": prepared.subset_path.name, + }, + "root_scope_id": root_scope, + "haystacks": [ + { + "digest": group.digest, + "scope_id": group_scopes[group.digest], + "trajectory_count": len(group.trajectory_ids), + "question_ids": list(group.question_ids), + } + for group in groups + ], + }, + ) + + targets: dict[str, list[str]] = {} + for group in groups: + for trajectory_id in group.trajectory_ids: + targets.setdefault(trajectory_id, []).append(group.digest) + found: set[str] = set() + inserted: dict[str, int] = {group.digest: 0 for group in groups} + group_failures: dict[str, dict[str, object]] = {} + for trajectory in _jsonl( + data_root / "trajectories.jsonl", + "trajectories.jsonl", + expected_digest=lock.file_digests["trajectories.jsonl"], + ): + trajectory_id = trajectory.get("id") + if not isinstance(trajectory_id, str) or trajectory_id not in targets: + continue + if trajectory_id in found: + raise RetrievalSmokeError(f"Selected trajectory appears more than once: {trajectory_id}") + found.add(trajectory_id) + for digest in targets[trajectory_id]: + if digest in group_failures: + continue + try: + adapters[digest].insert(trajectory) + inserted[digest] += 1 + except Exception as error: # noqa: BLE001 - one haystack failure must not stop the other haystack + failure = _failure( + failure_id=f"ingest-{digest[:16]}", + phase="ingest", + error=error, + haystack_digest=digest, + question_id=None, + ) + group_failures[digest] = failure + _append_json(failures_path, failure) + missing = sorted(set(targets) - found) + if missing: + raise RetrievalSmokeError(f"Selected trajectories were not found: {missing[:5]}") + + succeeded = 0 + failed = 0 + context_bytes_total = 0 + citation_count = 0 + for sequence, question in enumerate(questions, start=1): + group = question_groups[question.question_id] + scope_id = group_scopes[group.digest] + if group.digest in group_failures: + failed += 1 + _append_json( + results_path, + _result( + sequence=sequence, + question=question, + group=group, + scope_id=scope_id, + status="failed", + memory_context=[], + metadata=None, + failure_id=str(group_failures[group.digest]["failure_id"]), + ), + ) + continue + adapter = adapters[group.digest] + invocation_id = f"{normalized_run_id}:{question.question_id}" + adapter.set_query_context(query_invocation_id=invocation_id) + try: + memory_context = adapter.query(question.text, query_image=question.image) + metadata = adapter.post_query_hook( + query=question.text, + query_image=question.image, + memory_context=memory_context, + ) + _validate_context(memory_context) + result = _result( + sequence=sequence, + question=question, + group=group, + scope_id=scope_id, + status="succeeded", + memory_context=memory_context, + metadata=metadata, + failure_id=None, + ) + succeeded += 1 + context_bytes_total += sum(len(item["value"].encode()) for item in memory_context) + citations = result["citations"] + citation_count += len(citations) if isinstance(citations, list) else 0 + except Exception as error: # noqa: BLE001 - each query needs an independent classified result + failed += 1 + failure_id = f"query-{question.question_id}" + try: + # The adapter keeps the failed query's provenance (requested and actual + # search mode) even when the query itself failed; keep it in the result. + metadata = adapter.post_query_hook( + query=question.text, + query_image=question.image, + memory_context=[], + ) + except Exception: # noqa: BLE001 - failure provenance must never mask the original failure + metadata = None + _append_json( + failures_path, + _failure( + failure_id=failure_id, + phase="query", + error=error, + haystack_digest=group.digest, + question_id=question.question_id, + ), + ) + result = _result( + sequence=sequence, + question=question, + group=group, + scope_id=scope_id, + status="failed", + memory_context=[], + metadata=metadata, + failure_id=failure_id, + ) + finally: + adapter.clear_query_context() + _append_json(results_path, result) + + summary = { + "schema": RETRIEVAL_SUMMARY_SCHEMA, + "classification": "smoke-subset-retrieval-only", + "run_id": normalized_run_id, + "experiment_arm": arm_manifest_record(arm), + "completed_at": datetime.now(UTC).isoformat(), + "question_count": len(questions), + "succeeded": succeeded, + "failed": failed, + "haystack_count": len(groups), + "trajectory_count": len(found), + "inserted_trajectory_copies": sum(inserted.values()), + "context_bytes": context_bytes_total, + "citation_count": citation_count, + "elapsed_ms": round((time.perf_counter_ns() - started_ns) / 1_000_000, 3), + "accuracy": None, + "reader": None, + "judge": None, + } + _write_json_exclusive(summary_path, summary) + return RetrievalSmokeRun( + output_dir=output_dir, + manifest_path=retrieval_manifest_path, + results_path=results_path, + failures_path=failures_path, + summary_path=summary_path, + audit_path=audit_path, + ) + + +def _require_runtime(runtime: RetrievalSmokeRuntime, arm: ExperimentArm) -> None: + """Fail as a capability error before any ingestion when the arm cannot be executed.""" + + readiness = runtime.get_readiness() + if readiness.get("status") != "ready": + raise RetrievalSmokeError("PowerContext Server is not ready") + capabilities = runtime.get_capabilities() + modes = capabilities.get("search_modes") + if arm.retrieval_strategy == "memory-search": + if arm.search_mode is None or not isinstance(modes, list) or arm.search_mode not in modes: + raise RetrievalCapabilityError( + f"PowerContext Server does not support {arm.search_mode} Memory search " + f"required by experiment arm {arm.arm_id}" + ) + elif arm.retrieval_strategy == "prepared-context": + versions = capabilities.get("context_versions") + if not isinstance(versions, list) or "powercontext.prepared-context.v1" not in versions: + raise RetrievalCapabilityError( + f"PowerContext Server does not support PreparedContext required by experiment arm {arm.arm_id}" + ) + else: + raise RetrievalCapabilityError(f"unsupported retrieval strategy for experiment arm {arm.arm_id}") + source_types = capabilities.get("source_types") + if not isinstance(source_types, list) or "content" not in source_types: + raise RetrievalCapabilityError("PowerContext Server does not support Content Sources") + artifact_families = capabilities.get("artifact_families") + if not isinstance(artifact_families, list) or "memory" not in artifact_families: + raise RetrievalCapabilityError("PowerContext Server does not support Memory artifacts") + + +def _create_scope( + runtime: RetrievalSmokeRuntime, + *, + title: str, + summary: str, + idempotency_seed: str, + parent_scope_id: str | None, +) -> str: + response = runtime.create_scope( + { + "title": title, + "summary": summary, + "parent_scope_id": parent_scope_id, + "idempotency_key": hashlib.sha256(idempotency_seed.encode()).hexdigest(), + } + ) + return _nonblank(response.get("scope_id"), "created scope_id") + + +def _load_questions( + path: Path, + selection: SmokeSelection, + *, + expected_digest: str, +) -> tuple[RetrievalQuestion, ...]: + cases = {case.question_id: case for case in selection.cases} + found: dict[str, RetrievalQuestion] = {} + for row in _jsonl(path, "questions.jsonl", expected_digest=expected_digest): + question_id = row.get("id") + if not isinstance(question_id, str) or question_id not in cases: + continue + if question_id in found: + raise RetrievalSmokeError(f"Selected question appears more than once: {question_id}") + domain = row.get("domain") + if domain not in {"web", "enterprise"}: + raise RetrievalSmokeError(f"Selected question has invalid domain: {question_id}") + text, image = _question_components( + row.get("question"), + row.get("image"), + data_root=path.parent, + question_id=question_id, + ) + found[question_id] = RetrievalQuestion( + question_id=question_id, + domain=cast("Literal['web', 'enterprise']", domain), + ability=cases[question_id].ability, + text=text, + image=image, + ) + missing = [case.question_id for case in selection.cases if case.question_id not in found] + if missing: + raise RetrievalSmokeError(f"Selected questions were not found: {missing}") + return tuple(found[case.question_id] for case in selection.cases) + + +def _load_groups( + path: Path, + selection: SmokeSelection, + *, + expected_digest: str, +) -> tuple[tuple[HaystackGroup, ...], dict[str, HaystackGroup]]: + try: + content = path.read_bytes() + if hashlib.sha256(content).hexdigest() != expected_digest: + raise RetrievalSmokeError("LongMemEval-V2 SHA-256 mismatch for haystack") + value = json.loads(content) + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error: + raise RetrievalSmokeError(f"Cannot read LongMemEval-V2 haystack: {path}") from error + if not isinstance(value, dict): + raise RetrievalSmokeError("LongMemEval-V2 haystack must be an object") + ordered: list[str] = [] + grouped_ids: dict[str, tuple[str, ...]] = {} + grouped_questions: dict[str, list[str]] = {} + question_digest: dict[str, str] = {} + for case in selection.cases: + raw_ids = value.get(case.question_id) + if not isinstance(raw_ids, list) or not raw_ids or not all(isinstance(item, str) and item for item in raw_ids): + raise RetrievalSmokeError(f"Invalid haystack for selected question: {case.question_id}") + trajectory_ids = tuple(raw_ids) + digest = hashlib.sha256( + json.dumps(trajectory_ids, ensure_ascii=True, separators=(",", ":")).encode() + ).hexdigest() + if digest not in grouped_ids: + ordered.append(digest) + grouped_ids[digest] = trajectory_ids + grouped_questions[digest] = [] + elif grouped_ids[digest] != trajectory_ids: + raise RetrievalSmokeError("Haystack digest collision") + grouped_questions[digest].append(case.question_id) + question_digest[case.question_id] = digest + groups = tuple( + HaystackGroup( + digest=digest, + trajectory_ids=grouped_ids[digest], + question_ids=tuple(grouped_questions[digest]), + ) + for digest in ordered + ) + by_digest = {group.digest: group for group in groups} + return groups, {question_id: by_digest[digest] for question_id, digest in question_digest.items()} + + +def _question_components( + value: object, + raw_image: object, + *, + data_root: Path, + question_id: str, +) -> tuple[str, str | None]: + text: object = None + if isinstance(value, str) and value.strip(): + text = value + if isinstance(value, Mapping): + text = value.get("text") + raw_image = value.get("image", raw_image) + if not isinstance(text, str) or not text.strip(): + raise RetrievalSmokeError(f"Selected question has invalid text content: {question_id}") + if raw_image is None: + return text, None + if not isinstance(raw_image, str) or not raw_image.strip(): + raise RetrievalSmokeError(f"Selected question has invalid image content: {question_id}") + image_path = Path(raw_image) + if not image_path.is_absolute(): + image_path = data_root / image_path + if not image_path.is_file(): + raise RetrievalSmokeError(f"Selected question image is missing: {question_id}") + return text, str(image_path.resolve()) + + +def _jsonl(path: Path, label: str, *, expected_digest: str) -> Iterator[dict[str, object]]: + hasher = hashlib.sha256() + try: + with path.open("rb") as stream: + for line_number, raw_line in enumerate(stream, start=1): + hasher.update(raw_line) + if not raw_line.strip(): + continue + try: + value = json.loads(raw_line) + except (UnicodeDecodeError, json.JSONDecodeError) as error: + raise RetrievalSmokeError(f"{label} has invalid JSON at line {line_number}") from error + if not isinstance(value, dict): + raise RetrievalSmokeError(f"{label} line {line_number} must be an object") + yield value + except OSError as error: + raise RetrievalSmokeError(f"Cannot stream {label}: {path}") from error + if hasher.hexdigest() != expected_digest: + raise RetrievalSmokeError(f"LongMemEval-V2 SHA-256 mismatch while streaming {label}") + + +def _result( + *, + sequence: int, + question: RetrievalQuestion, + group: HaystackGroup, + scope_id: str, + status: Literal["succeeded", "failed"], + memory_context: list[dict[str, str]], + metadata: dict[str, object] | None, + failure_id: str | None, +) -> dict[str, object]: + context_bytes = sum(len(item["value"].encode()) for item in memory_context) + citations = [] if metadata is None else metadata.get("citations", []) + timings = None if metadata is None else metadata.get("timings_ms") + return { + "schema": RETRIEVAL_RESULT_SCHEMA, + "sequence": sequence, + "question_id": question.question_id, + "domain": question.domain, + "ability": question.ability, + "question": {"text": question.text, "image": question.image}, + "haystack_digest": group.digest, + "scope_id": scope_id, + "status": status, + "search_mode": { + "requested": None if metadata is None else metadata.get("requested_mode"), + "actual": None if metadata is None else metadata.get("actual_mode"), + }, + "memory_context": memory_context, + "context_bytes": context_bytes, + "citations": citations, + "timings_ms": timings, + "failure_id": failure_id, + } + + +def _failure( + *, + failure_id: str, + phase: Literal["ingest", "query"], + error: Exception, + haystack_digest: str, + question_id: str | None, +) -> dict[str, object]: + summary = str(error).strip() or type(error).__name__ + return { + "schema": RETRIEVAL_FAILURE_SCHEMA, + "failure_id": failure_id, + "phase": phase, + "category": _failure_category(error), + "error_type": type(error).__name__, + "summary": summary[:500], + "haystack_digest": haystack_digest, + "question_id": question_id, + } + + +def _failure_category(error: Exception) -> Literal["infrastructure", "integration", "integrity"]: + if isinstance(error, PowerContextMemoryModeError): + return "integrity" + if isinstance(error, PowerContextMemoryAdapterError): + return "infrastructure" + return "integration" + + +def _validate_context(items: list[dict[str, str]]) -> None: + if not isinstance(items, list): + raise RetrievalSmokeError("Memory query did not return a list") + for index, item in enumerate(items): + if set(item) != {"type", "value"} or item["type"] not in {"text", "image"} or not item["value"]: + raise RetrievalSmokeError(f"Memory context item {index} is invalid") + + +def _create_empty(path: Path) -> None: + with path.open("x", encoding="utf-8"): + pass + + +def _append_json(path: Path, value: object) -> None: + with path.open("a", encoding="utf-8", newline="") as stream: + stream.write(json.dumps(value, ensure_ascii=False, separators=(",", ":"), allow_nan=False) + "\n") + + +def _write_json_exclusive(path: Path, value: object) -> None: + with path.open("x", encoding="utf-8", newline="") as stream: + stream.write(json.dumps(value, ensure_ascii=True, indent=2, sort_keys=True, allow_nan=False) + "\n") + + +def _nonblank(value: object, label: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise RetrievalSmokeError(f"{label} must be a non-empty string") + return value.strip() diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/run_smoke.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/run_smoke.py new file mode 100644 index 0000000000..ea110fdbbf --- /dev/null +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/run_smoke.py @@ -0,0 +1,699 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Run every LongMemEval-V2 smoke stage into one fail-closed run directory.""" + +from __future__ import annotations + +import hashlib +import json +import os +import time +from collections.abc import Callable, Mapping +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path +from typing import Literal, TypeAlias +from urllib.parse import urlsplit, urlunsplit + +from powercontext_eval.benchmarks.longmemeval_v2.adapter import ( + PowerContextMemoryAdapterError, + PowerContextMemoryModeError, +) +from powercontext_eval.benchmarks.longmemeval_v2.arms import ( + ExperimentArm, + arm_manifest_record, + resolve_experiment_arm, +) +from powercontext_eval.benchmarks.longmemeval_v2.catalog import ( + UPSTREAM_HARNESS_COMMIT, + LongMemEvalV2CatalogError, + LongMemEvalV2EnvironmentError, + LongMemEvalV2InputError, +) +from powercontext_eval.benchmarks.longmemeval_v2.costs import ( + ModelPricePolicy, + cost_policy_record, +) +from powercontext_eval.benchmarks.longmemeval_v2.prepare_smoke import ( + DEFAULT_PROCESSOR_MODEL, + PreparedPromptRun, + PrepareSmokeError, + prepare_reader_inputs_smoke, +) +from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import ( + DEFAULT_ANTHROPIC_BASE_URL_ENV, + DEFAULT_ANTHROPIC_TOKEN_ENV, + DEFAULT_DEEPSEEK_BASE_URL, + DEFAULT_DEEPSEEK_MODEL, + DEFAULT_DEEPSEEK_TOKEN_ENV, + DEFAULT_READER_MODEL, + ReaderSmokeError, + ReaderSmokeRun, + ReaderTransport, + run_reader_smoke, +) +from powercontext_eval.benchmarks.longmemeval_v2.replay_score import ( + ReplayScoreError, + ReplayScoreRun, + replay_score_smoke, +) +from powercontext_eval.benchmarks.longmemeval_v2.report import build_report +from powercontext_eval.benchmarks.longmemeval_v2.retrieval_smoke import ( + RetrievalCapabilityError, + RetrievalSmokeError, + RetrievalSmokeRun, + RetrievalSmokeRuntime, + run_retrieval_smoke, +) +from powercontext_eval.benchmarks.longmemeval_v2.score_smoke import ScoreSmokeError, ScoreSmokeRun, run_score_smoke +from powercontext_eval.benchmarks.longmemeval_v2.smoke import PreparedSmokeRun, prepare_smoke_run +from powercontext_eval.errors import PowerContextEvalError + +RUN_MANIFEST_SCHEMA = "powercontext.longmemeval-v2-smoke-run.v1" +RUN_SUMMARY_SCHEMA = "powercontext.longmemeval-v2-smoke-run-summary.v1" +RUN_FAILURE_SCHEMA = "powercontext.longmemeval-v2-smoke-run-failure.v1" + +Phase: TypeAlias = Literal["preflight", "retrieval", "prepare", "reader", "score", "replay"] +ErrorClass: TypeAlias = Literal["configuration", "infrastructure", "retrieval", "generation", "judge", "integrity"] +RunStatus: TypeAlias = Literal["completed", "partial", "failed"] + +PHASES: tuple[Phase, ...] = ("preflight", "retrieval", "prepare", "reader", "score", "replay") +PHASE_DIRECTORIES: Mapping[Phase, str] = { + "preflight": "01-inputs", + "retrieval": "02-retrieval", + "prepare": "03-prepare", + "reader": "04-reader", + "score": "05-score", + "replay": "06-replay", +} + + +class RunSmokeError(PowerContextEvalError): + """The one-command smoke run cannot preserve its evaluation contract.""" + + +@dataclass(frozen=True) +class SmokeStages: + """The reusable stage entry points, injectable so tests never reach a model or server.""" + + preflight: Callable[..., PreparedSmokeRun] = prepare_smoke_run + retrieval: Callable[..., RetrievalSmokeRun] = run_retrieval_smoke + prepare: Callable[..., PreparedPromptRun] = prepare_reader_inputs_smoke + reader: Callable[..., ReaderSmokeRun] = run_reader_smoke + score: Callable[..., ScoreSmokeRun] = run_score_smoke + replay: Callable[..., ReplayScoreRun] = replay_score_smoke + + +@dataclass(frozen=True) +class SmokeRunResult: + """The inspectable outcome of one orchestrated smoke run.""" + + output_dir: Path + manifest_path: Path + summary_path: Path + failures_path: Path + status: RunStatus + completed_phases: tuple[Phase, ...] + skipped_phases: tuple[Phase, ...] + failed_phase: Phase | None + + +def classify_error(error: BaseException) -> ErrorClass: + """Map one raised error to the failure class recorded for review.""" + + if isinstance(error, RunSmokeError): + return "configuration" + if isinstance(error, LongMemEvalV2InputError): + return "configuration" + if isinstance(error, LongMemEvalV2EnvironmentError): + return "infrastructure" + if isinstance(error, LongMemEvalV2CatalogError): + return "integrity" + if isinstance(error, PowerContextMemoryModeError): + return "integrity" + if isinstance(error, RetrievalCapabilityError): + return "infrastructure" + if isinstance(error, PowerContextMemoryAdapterError): + return "infrastructure" + if isinstance(error, PrepareSmokeError): + return "infrastructure" + if isinstance(error, ReaderSmokeError): + return "generation" + if isinstance(error, ScoreSmokeError): + return "judge" + if isinstance(error, ReplayScoreError): + return "integrity" + if isinstance(error, RetrievalSmokeError): + return "retrieval" + return "infrastructure" + + +def run_smoke( + *, + data_root: Path, + dataset_lock: Path, + smoke_manifest: Path, + harness_root: Path, + harness_python: Path, + processor_revision: str, + output_dir: Path, + powercontext_revision: str, + integration_revision: str, + run_id: str | None = None, + processor_model: str = DEFAULT_PROCESSOR_MODEL, + memory_context_max_tokens: int = 200_000, + powercontext_base_url: str = "http://127.0.0.1:8000", + powercontext_token_env: str = "POWERCONTEXT_TOKEN", + experiment_arm: str | ExperimentArm | None = None, + search_limit: int = 10, + timeout_seconds: float = 30.0, + reader_provider: str = "deepseek-openai", + reader_model: str | None = None, + reader_base_url: str | None = None, + reader_base_url_env: str = DEFAULT_ANTHROPIC_BASE_URL_ENV, + reader_token_env: str | None = None, + reader_max_tokens: int = 512, + reader_temperature: float = 0.0, + reader_timeout_seconds: float = 120.0, + judge_provider: str = "deepseek-openai", + judge_model: str = DEFAULT_DEEPSEEK_MODEL, + judge_token_env: str = DEFAULT_DEEPSEEK_TOKEN_ENV, + judge_base_url: str = DEFAULT_DEEPSEEK_BASE_URL, + judge_max_tokens: int = 256, + judge_temperature: float = 0.0, + judge_timeout_seconds: float = 120.0, + reader_price_policy: ModelPricePolicy | None = None, + judge_price_policy: ModelPricePolicy | None = None, + skip_reader: bool = False, + skip_score: bool = False, + stages: SmokeStages | None = None, + runtime: RetrievalSmokeRuntime | None = None, + reader_transport: ReaderTransport | None = None, + judge_transport: ReaderTransport | None = None, +) -> SmokeRunResult: + """Run preflight, retrieval, prepare, Reader, Judge scoring, and replay into one run directory.""" + + active = stages or SmokeStages() + normalized_run_id = _nonblank(run_id or output_dir.name, "run_id") + powercontext_ref = _nonblank(powercontext_revision, "powercontext_revision") + integration_ref = _nonblank(integration_revision, "integration_revision") + processor_ref = _nonblank(processor_revision, "processor_revision") + processor_name = _nonblank(processor_model, "processor_model") + arm = resolve_experiment_arm(experiment_arm) + if skip_reader: + skip_score = True + if reader_provider not in {"anthropic-compatible", "deepseek-openai"}: + raise RunSmokeError("reader_provider must be anthropic-compatible or deepseek-openai") + if judge_provider != "deepseek-openai": + raise RunSmokeError("judge_provider must be deepseek-openai") + if isinstance(search_limit, bool) or not 1 <= search_limit <= 50: + raise RunSmokeError("search_limit must be from 1 through 50") + if isinstance(memory_context_max_tokens, bool) or memory_context_max_tokens <= 0: + raise RunSmokeError("memory_context_max_tokens must be positive") + if output_dir.exists(): + raise RunSmokeError(f"Refusing to overwrite smoke run artifacts: {output_dir}") + + resolved_reader_token_env = _nonblank( + reader_token_env + or (DEFAULT_DEEPSEEK_TOKEN_ENV if reader_provider == "deepseek-openai" else None) + or DEFAULT_ANTHROPIC_TOKEN_ENV, + "reader_token_env", + ) + resolved_reader_model = _nonblank( + reader_model or (DEFAULT_DEEPSEEK_MODEL if reader_provider == "deepseek-openai" else DEFAULT_READER_MODEL), + "reader_model", + ) + resolved_judge_model = _nonblank(judge_model, "judge_model") + resolved_judge_token_env = _nonblank(judge_token_env, "judge_token_env") + resolved_powercontext_token_env = _nonblank(powercontext_token_env, "powercontext_token_env") + base_url = _credential_free_url(powercontext_base_url, "powercontext_base_url") + # The retrieval stage always resolves the PowerContext token, even in skip-reader mode, + # so its value must always be collected for redaction; Reader/Judge values are collected + # only while their stages can still raise errors that quote them. + secret_env_names = [resolved_powercontext_token_env] + if not skip_reader: + secret_env_names.append(resolved_reader_token_env) + if not skip_score: + secret_env_names.append(resolved_judge_token_env) + secrets = _known_secrets(tuple(secret_env_names)) + + lock_digest = _file_digest(dataset_lock, "dataset lock") + manifest_digest = _file_digest(smoke_manifest, "smoke manifest") + started_at = datetime.now(UTC) + started_ns = time.perf_counter_ns() + try: + output_dir.mkdir(parents=True, exist_ok=False) + except FileExistsError as error: + raise RunSmokeError(f"Refusing to overwrite smoke run artifacts: {output_dir}") from error + except OSError as error: + raise RunSmokeError(f"Cannot create smoke run directory: {output_dir}") from error + + manifest_path = output_dir / "run-manifest.json" + summary_path = output_dir / "run-summary.json" + failures_path = output_dir / "failures.jsonl" + outputs = {phase: output_dir / PHASE_DIRECTORIES[phase] for phase in PHASES} + with failures_path.open("x", encoding="utf-8", newline=""): + pass + _write_json_exclusive( + manifest_path, + { + "schema": RUN_MANIFEST_SCHEMA, + "classification": "smoke-subset", + "run_id": normalized_run_id, + "started_at": started_at.isoformat(), + "phases": {phase: PHASE_DIRECTORIES[phase] for phase in PHASES}, + "modes": {"reader": not skip_reader, "score": not skip_score}, + "experiment_arm": arm_manifest_record(arm), + "inputs": { + "data_root": str(data_root.resolve()), + "dataset_lock": {"path": str(dataset_lock.resolve()), "content_sha256": lock_digest}, + "smoke_manifest": {"path": str(smoke_manifest.resolve()), "content_sha256": manifest_digest}, + "harness": { + "root": str(harness_root.resolve()), + "commit": UPSTREAM_HARNESS_COMMIT, + "python": str(harness_python.resolve()), + }, + }, + "processor": {"model": processor_name, "revision": processor_ref}, + "memory_context_max_tokens": memory_context_max_tokens, + "powercontext": { + "base_url": base_url, + "token_env": resolved_powercontext_token_env, + "search_mode": arm.search_mode, + "search_limit": search_limit, + "timeout_seconds": timeout_seconds, + }, + "reader": None + if skip_reader + else { + "provider": reader_provider, + "model": resolved_reader_model, + "token_env": resolved_reader_token_env, + "base_url": _reader_base_url_record(reader_provider, reader_base_url, reader_base_url_env), + "max_tokens": reader_max_tokens, + "temperature": reader_temperature, + "timeout_seconds": reader_timeout_seconds, + }, + "judge": None + if skip_score + else { + "provider": judge_provider, + "model": resolved_judge_model, + "token_env": resolved_judge_token_env, + "base_url": _credential_free_url(judge_base_url, "judge_base_url"), + "max_tokens": judge_max_tokens, + "temperature": judge_temperature, + "timeout_seconds": judge_timeout_seconds, + }, + "revisions": {"powercontext": powercontext_ref, "integration": integration_ref}, + "cost_policy": { + "reader": cost_policy_record(reader_price_policy), + "judge": cost_policy_record(judge_price_policy), + }, + "privacy": { + "credentials": "resolved-from-environment-at-runtime-never-recorded", + "reference_answers": "never-read-by-the-adapter-retrieval-prepare-or-reader-stages", + "reference_answers_readers": ( + "the score stage reads the locked local questions to score and to write the local replay artifact; " + "the replay stage reads only that local artifact" + ), + "reference_answers_artifact": "05-score/scoring-inputs.local.jsonl", + }, + }, + ) + + completed: list[Phase] = [] + failures: list[dict[str, object]] = [] + failed: Phase | None = None + + def persist(status: RunStatus, failed_phase: Phase | None) -> None: + _write_summary( + summary_path, + output_dir=output_dir, + completed=completed, + skipped=tuple(phase for phase in PHASES if phase not in completed), + failures=failures, + status=status, + failed_phase=failed_phase, + elapsed_ms=round((time.perf_counter_ns() - started_ns) / 1_000_000, 3), + ) + + def run_phase(phase: Phase, action: Callable[[], object]) -> bool: + nonlocal failed + try: + action() + except Exception as error: # noqa: BLE001 - every phase failure needs an independent classification + failure: dict[str, object] = { + "schema": RUN_FAILURE_SCHEMA, + "phase": phase, + "error_class": classify_error(error), + "error_type": type(error).__name__, + "summary": _redact(str(error).strip() or type(error).__name__, secrets)[:500], + } + _append_json(failures_path, failure) + failures.append(failure) + failed = phase + persist("failed", phase) + return False + completed.append(phase) + persist(_in_progress_status(completed, skip_reader, skip_score), None) + return True + + def finish() -> SmokeRunResult: + skipped = tuple(phase for phase in PHASES if phase not in completed) + if failed is not None: + status: RunStatus = "failed" + elif skipped: + status = "partial" + else: + status = "completed" + persist(status, failed) + build_report(run_dir=output_dir) + return SmokeRunResult( + output_dir=output_dir, + manifest_path=manifest_path, + summary_path=summary_path, + failures_path=failures_path, + status=status, + completed_phases=tuple(completed), + skipped_phases=skipped, + failed_phase=failed, + ) + + def preflight() -> None: + _require_reader_configuration( + enabled=not skip_reader, + provider=reader_provider, + token_env=resolved_reader_token_env, + base_url=reader_base_url, + base_url_env=reader_base_url_env, + transport=reader_transport, + ) + _require_token( + enabled=not skip_score, token_env=resolved_judge_token_env, label="Judge", transport=judge_transport + ) + active.preflight( + data_root=data_root, + dataset_lock=dataset_lock, + harness_root=harness_root, + smoke_manifest=smoke_manifest, + output_dir=outputs["preflight"], + ) + + if not run_phase("preflight", preflight): + return finish() + if not run_phase( + "retrieval", + lambda: active.retrieval( + data_root=data_root, + dataset_lock=dataset_lock, + harness_root=harness_root, + smoke_manifest=smoke_manifest, + output_dir=outputs["retrieval"], + run_id=normalized_run_id, + powercontext_revision=powercontext_ref, + integration_revision=integration_ref, + base_url=powercontext_base_url, + token_env=resolved_powercontext_token_env, + experiment_arm=arm, + search_limit=search_limit, + timeout_seconds=timeout_seconds, + runtime=runtime, + ), + ): + return finish() + if not run_phase( + "prepare", + lambda: active.prepare( + retrieval_dir=outputs["retrieval"], + harness_root=harness_root, + harness_python=harness_python, + output_dir=outputs["prepare"], + processor_model=processor_name, + processor_revision=processor_ref, + memory_context_max_tokens=memory_context_max_tokens, + ), + ): + return finish() + if skip_reader: + return finish() + if not run_phase( + "reader", + lambda: active.reader( + prepared_dir=outputs["prepare"], + output_dir=outputs["reader"], + provider=reader_provider, + model=resolved_reader_model, + base_url=reader_base_url, + base_url_env=reader_base_url_env, + token_env=resolved_reader_token_env, + max_tokens=reader_max_tokens, + temperature=reader_temperature, + timeout_seconds=reader_timeout_seconds, + price_policy=reader_price_policy, + transport=reader_transport, + ), + ): + return finish() + if skip_score: + return finish() + if not run_phase( + "score", + lambda: active.score( + reader_dir=outputs["reader"], + data_root=data_root, + dataset_lock=dataset_lock, + smoke_manifest=smoke_manifest, + harness_root=harness_root, + output_dir=outputs["score"], + judge_model=resolved_judge_model, + judge_token_env=resolved_judge_token_env, + judge_base_url=judge_base_url, + judge_max_tokens=judge_max_tokens, + judge_temperature=judge_temperature, + judge_timeout_seconds=judge_timeout_seconds, + judge_price_policy=judge_price_policy, + judge_transport=judge_transport, + ), + ): + return finish() + if not run_phase( + "replay", + lambda: active.replay( + score_dir=outputs["score"], + harness_root=harness_root, + output_dir=outputs["replay"], + ), + ): + return finish() + return finish() + + +def _write_summary( + summary_path: Path, + *, + output_dir: Path, + completed: list[Phase], + skipped: tuple[Phase, ...], + failures: list[dict[str, object]], + status: RunStatus, + failed_phase: Phase | None, + elapsed_ms: float, +) -> None: + _write_json_replace( + summary_path, + { + "schema": RUN_SUMMARY_SCHEMA, + "classification": "smoke-subset", + "status": status, + "completed_at": datetime.now(UTC).isoformat(), + "completed_phases": list(completed), + "skipped_phases": list(skipped), + "failed_phase": failed_phase, + "question_count": _question_count(output_dir), + "accuracy": _accuracy(output_dir), + "failures": list(failures), + "artifacts": _artifacts(output_dir), + "elapsed_ms": elapsed_ms, + }, + ) + + +def _in_progress_status(completed: list[Phase], skip_reader: bool, skip_score: bool) -> RunStatus: + """Report an in-progress run without claiming phases that are still pending. + + A run may read as ``completed`` only after every phase ran, so a skip-reader or + skip-score run stays ``partial`` and an abnormal exit cannot leave a ``completed`` + summary behind while retrieval or prepare has not executed. + """ + + if skip_reader or skip_score or any(phase not in completed for phase in PHASES): + return "partial" + return "completed" + + +def _artifacts(output_dir: Path) -> dict[str, str | None]: + names = {"run_manifest": "run-manifest.json", "run_summary": "run-summary.json", "failures": "failures.jsonl"} + names.update({phase: PHASE_DIRECTORIES[phase] for phase in PHASES}) + return {name: value if (output_dir / value).exists() else None for name, value in names.items()} + + +def _question_count(output_dir: Path) -> int | None: + for phase in ("score", "replay", "reader", "prepare", "retrieval"): + summary = _load_optional_json(output_dir / PHASE_DIRECTORIES[phase] / _summary_name(phase)) + if summary is None: + continue + count = summary.get("question_count") + if isinstance(count, int) and not isinstance(count, bool): + return count + return None + + +def _accuracy(output_dir: Path) -> dict[str, object] | None: + for phase in ("score", "replay"): + summary = _load_optional_json(output_dir / PHASE_DIRECTORIES[phase] / _summary_name(phase)) + if summary is None or summary.get("failed") != 0: + continue + correct = summary.get("correct") + incorrect = summary.get("incorrect") + total = summary.get("question_count") + value = summary.get("accuracy") + if isinstance(correct, int) and isinstance(incorrect, int) and isinstance(total, int): + return { + "correct": correct, + "incorrect": incorrect, + "failed": 0, + "value": value if isinstance(value, float) else None, + } + return None + + +def _summary_name(phase: Phase) -> str: + return { + "retrieval": "summary.json", + "prepare": "prepare-summary.json", + "reader": "reader-summary.json", + "score": "score-summary.json", + "replay": "replay-summary.json", + }[phase] + + +def _load_optional_json(path: Path) -> dict[str, object] | None: + try: + value = json.loads(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError): + return None + return value if isinstance(value, dict) else None + + +def _require_reader_configuration( + *, + enabled: bool, + provider: str, + token_env: str, + base_url: str | None, + base_url_env: str, + transport: ReaderTransport | None, +) -> None: + if not enabled or transport is not None: + return + if provider == "anthropic-compatible" and base_url is None: + _require_token(enabled=True, token_env=base_url_env, label="Reader base URL", transport=None) + _require_token(enabled=True, token_env=token_env, label="Reader", transport=None) + + +def _require_token(*, enabled: bool, token_env: str, label: str, transport: ReaderTransport | None) -> None: + """Fail as a configuration error before a model-backed stage spends any work.""" + + if not enabled or transport is not None: + return + if not os.getenv(token_env, "").strip(): + raise RunSmokeError(f"{label} requires the {token_env} environment variable") + + +def _known_secrets(env_names: tuple[str, ...]) -> tuple[str, ...]: + """Collect the current secret values used only to redact failure summaries, never recorded.""" + + values = (os.getenv(name, "") for name in env_names) + return tuple(value for value in values if value) + + +def _redact(summary: str, secrets: tuple[str, ...]) -> str: + for secret in secrets: + summary = summary.replace(secret, "") + return summary + + +def _reader_base_url_record(provider: str, base_url: str | None, base_url_env: str) -> str: + if base_url is not None: + return _credential_free_url(base_url, "reader_base_url") + if provider == "deepseek-openai": + return DEFAULT_DEEPSEEK_BASE_URL + return f"environment:{base_url_env}" + + +def _credential_free_url(value: str, label: str) -> str: + parsed = urlsplit(_nonblank(value, label)) + if not parsed.scheme or not parsed.hostname: + raise RunSmokeError(f"{label} must be an absolute URL") + if parsed.username is not None or parsed.password is not None: + raise RunSmokeError(f"{label} must not contain credentials") + if parsed.query or parsed.fragment: + raise RunSmokeError(f"{label} must not contain query or fragment data") + host = f"[{parsed.hostname}]" if ":" in parsed.hostname else parsed.hostname + netloc = f"{host}:{parsed.port}" if parsed.port is not None else host + return urlunsplit((parsed.scheme, netloc, parsed.path.rstrip("/"), "", "")) + + +def _file_digest(path: Path, label: str) -> str: + hasher = hashlib.sha256() + try: + with path.open("rb") as stream: + while chunk := stream.read(1024 * 1024): + hasher.update(chunk) + except OSError as error: + raise RunSmokeError(f"Cannot read LongMemEval-V2 {label}: {path}") from error + return hasher.hexdigest() + + +def _write_json_exclusive(path: Path, value: object) -> None: + try: + with path.open("x", encoding="utf-8", newline="") as stream: + stream.write(json.dumps(value, ensure_ascii=True, indent=2, sort_keys=True, allow_nan=False) + "\n") + except OSError as error: + raise RunSmokeError(f"Cannot write smoke run artifact: {path}") from error + + +def _write_json_replace(path: Path, value: object) -> None: + """Update the run summary in place without ever publishing a partial document.""" + + temporary = path.with_name(f".{path.name}.partial") + try: + with temporary.open("w", encoding="utf-8", newline="") as stream: + stream.write(json.dumps(value, ensure_ascii=True, indent=2, sort_keys=True, allow_nan=False) + "\n") + os.replace(temporary, path) + except OSError as error: + raise RunSmokeError(f"Cannot update smoke run summary: {path}") from error + + +def _append_json(path: Path, value: object) -> None: + with path.open("a", encoding="utf-8", newline="") as stream: + stream.write(json.dumps(value, ensure_ascii=False, separators=(",", ":"), allow_nan=False) + "\n") + + +def _nonblank(value: object, label: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise RunSmokeError(f"{label} must be a non-empty string") + return value.strip() diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/score_smoke.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/score_smoke.py new file mode 100644 index 0000000000..8fcefb5808 --- /dev/null +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/score_smoke.py @@ -0,0 +1,567 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Score LongMemEval-V2 Reader outputs with pinned rules and an optional DeepSeek Judge.""" + +from __future__ import annotations + +import hashlib +import importlib +import json +import os +import sys +import time +from collections.abc import Iterator, Mapping +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path +from typing import Any, Protocol, cast + +from powercontext_eval.benchmarks.longmemeval_v2.catalog import ( + LongMemEvalV2Catalog, + SmokeSelection, + load_dataset_lock, + load_smoke_manifest, + validate_harness_checkout, +) +from powercontext_eval.benchmarks.longmemeval_v2.costs import ( + ModelPricePolicy, + UsageAccount, + cost_policy_record, + usage_cost_block, +) +from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import ( + DEFAULT_DEEPSEEK_BASE_URL, + DEFAULT_DEEPSEEK_MODEL, + DEFAULT_DEEPSEEK_TOKEN_ENV, + DeepSeekOpenAIReader, + ReaderResponseError, + ReaderTransport, +) +from powercontext_eval.errors import PowerContextEvalError + +SCORE_MANIFEST_SCHEMA = "powercontext.longmemeval-v2-score-run.v1" +SCORE_INPUT_SCHEMA = "powercontext.longmemeval-v2-score-input.v1" +PER_QUESTION_SCHEMA = "powercontext.longmemeval-v2-score-result.v1" +JUDGE_OUTPUT_SCHEMA = "powercontext.longmemeval-v2-judge-output.v1" +SCORE_FAILURE_SCHEMA = "powercontext.longmemeval-v2-score-failure.v1" +SCORE_SUMMARY_SCHEMA = "powercontext.longmemeval-v2-score-summary.v1" +LLM_JUDGE_NAMES = {"llm_abstention_checker", "llm_gotchas_checker"} + + +class ScoreSmokeError(PowerContextEvalError): + """Reader outputs cannot be scored under the pinned LongMemEval-V2 contract.""" + + +class JudgeJudgementError(ScoreSmokeError): + """A completed Judge call could not be parsed into a judgement; its usage evidence is preserved.""" + + def __init__(self, message: str, *, usage: dict[str, object], judge_latency_ms: float) -> None: + super().__init__(message) + self.usage = usage + self.judge_latency_ms = judge_latency_ms + + +class MetricsAPI(Protocol): + def eval_name(self, eval_spec: str) -> str: ... + + def extract_boxed_answer(self, text: str) -> str: ... + + def eval_from_spec(self, spec: str, *args: Any, **kwargs: Any) -> Any: ... + + def score_to_bool(self, value: Any) -> bool: ... + + def _build_abstention_judge_messages(self, **kwargs: Any) -> list[dict[str, str]]: ... + + def _build_gotchas_judge_messages(self, **kwargs: Any) -> list[dict[str, str]]: ... + + def _parse_llm_binary_judgement(self, text: str) -> tuple[int, str]: ... + + +@dataclass(frozen=True) +class ScoreSmokeRun: + output_dir: Path + manifest_path: Path + inputs_path: Path + results_path: Path + judge_outputs_path: Path + failures_path: Path + summary_path: Path + + +def run_score_smoke( + *, + reader_dir: Path, + data_root: Path, + dataset_lock: Path, + smoke_manifest: Path, + harness_root: Path, + output_dir: Path, + judge_model: str = DEFAULT_DEEPSEEK_MODEL, + judge_token_env: str = DEFAULT_DEEPSEEK_TOKEN_ENV, + judge_base_url: str = DEFAULT_DEEPSEEK_BASE_URL, + judge_max_tokens: int = 256, + judge_temperature: float = 0.0, + judge_timeout_seconds: float = 120.0, + judge_price_policy: ModelPricePolicy | None = None, + judge_transport: ReaderTransport | None = None, +) -> ScoreSmokeRun: + """Score ten Reader answers and call an LLM Judge only for upstream LLM metric types.""" + + if output_dir.exists(): + raise ScoreSmokeError(f"Refusing to overwrite score artifacts: {output_dir}") + validate_harness_checkout(harness_root) + lock = load_dataset_lock(dataset_lock) + declared_selection = load_smoke_manifest(smoke_manifest) + if declared_selection.tier != lock.tier: + raise ScoreSmokeError("Smoke manifest tier does not match the dataset lock") + catalog = LongMemEvalV2Catalog.load( + data_root, + tier=declared_selection.tier, + expected_digests=lock.file_digests, + ) + selection = catalog.select_smoke(declared_selection.cases) + reader_manifest = _load_json(reader_dir / "reader-manifest.json", "reader manifest") + reader_summary = _load_json(reader_dir / "reader-summary.json", "reader summary") + outputs_path = reader_dir / "reader-outputs.jsonl" + if not outputs_path.is_file(): + raise ScoreSmokeError(f"Missing Reader outputs: {outputs_path}") + _validate_reader_artifacts(reader_manifest, reader_summary) + outputs = _reader_outputs(outputs_path, selection) + + metrics = _load_metrics(harness_root) + if judge_transport is None: + token = _nonblank(os.getenv(judge_token_env), judge_token_env) + transport: ReaderTransport = DeepSeekOpenAIReader( + judge_base_url, + token=token, + model=_nonblank(judge_model, "judge_model"), + max_tokens=judge_max_tokens, + temperature=judge_temperature, + timeout_seconds=judge_timeout_seconds, + ) + else: + transport = judge_transport + + questions = _selected_questions(data_root / "questions.jsonl", selection) + try: + output_dir.mkdir(parents=True, exist_ok=False) + except FileExistsError as error: + raise ScoreSmokeError(f"Refusing to overwrite score artifacts: {output_dir}") from error + except OSError as error: + raise ScoreSmokeError(f"Cannot create score artifact directory: {output_dir}") from error + + manifest_path = output_dir / "score-manifest.json" + inputs_path = output_dir / "scoring-inputs.local.jsonl" + results_path = output_dir / "per-question.jsonl" + judge_outputs_path = output_dir / "judge-outputs.jsonl" + failures_path = output_dir / "score-failures.jsonl" + summary_path = output_dir / "score-summary.json" + _write_json_exclusive( + manifest_path, + { + "schema": SCORE_MANIFEST_SCHEMA, + "classification": "smoke-subset-score-only", + "reader": { + "directory": str(reader_dir.resolve()), + "manifest_sha256": _file_digest(reader_dir / "reader-manifest.json"), + "outputs_sha256": _file_digest(outputs_path), + "summary_sha256": _file_digest(reader_dir / "reader-summary.json"), + }, + "judge": { + "provider": "deepseek-openai", + "model": judge_model, + "token_env": judge_token_env, + "base_url": "configured-directly", + "max_tokens": judge_max_tokens, + "temperature": judge_temperature, + }, + "cost_policy": cost_policy_record(judge_price_policy), + }, + ) + _create_empty(inputs_path) + _create_empty(results_path) + _create_empty(judge_outputs_path) + _create_empty(failures_path) + + started_ns = time.perf_counter_ns() + correct = 0 + failed = 0 + judge_account = UsageAccount() + for sequence, question in enumerate(questions, start=1): + question_id = _nonblank(question.get("id"), "question.id") + reader = outputs[question_id] + try: + score_input, result, judge_output = _score_one( + metrics, question, reader, sequence=sequence, transport=transport + ) + except JudgeJudgementError as error: + # The Judge transport already completed and returned usage; count the call + # and keep the evidence with the failure instead of losing the accounting. + failed += 1 + judge_account.account(error.usage) + _append_json( + failures_path, + { + "schema": SCORE_FAILURE_SCHEMA, + "question_id": question["id"], + "phase": "scoring", + "error_type": type(error).__name__, + "summary": (str(error).strip() or type(error).__name__)[:500], + "judge_usage": error.usage, + "judge_latency_ms": error.judge_latency_ms, + }, + ) + continue + except Exception as error: # noqa: BLE001 - one scoring failure must not hide subsequent outcomes + failed += 1 + _append_json( + failures_path, + { + "schema": SCORE_FAILURE_SCHEMA, + "question_id": question["id"], + "phase": "scoring", + "error_type": type(error).__name__, + "summary": (str(error).strip() or type(error).__name__)[:500], + }, + ) + continue + _append_json(inputs_path, score_input) + _append_json(results_path, result) + if judge_output is not None: + _append_json(judge_outputs_path, judge_output) + judge_account.account(judge_output["usage"]) + result_correct = result["correct"] + if not isinstance(result_correct, bool): + raise TypeError("score result correct field must be a boolean") + correct += int(result_correct) + total = len(questions) + _write_json_exclusive( + summary_path, + { + "schema": SCORE_SUMMARY_SCHEMA, + "classification": "smoke-subset-score-only", + "completed_at": datetime.now(UTC).isoformat(), + "question_count": total, + "correct": correct, + "incorrect": total - correct - failed, + "failed": failed, + "accuracy": None if failed else correct / total, + "judge_calls": judge_account.calls, + "judge_usage": _judge_usage_totals( + input_tokens=judge_account.input_tokens, + output_tokens=judge_account.output_tokens, + cache_hit_tokens=judge_account.cache_hit_tokens, + cache_miss_tokens=judge_account.cache_miss_tokens, + cache_split_reported=judge_account.cache_split_reported, + ), + "judge_cost": usage_cost_block( + judge_price_policy, + provider="deepseek-openai", + model=_nonblank(judge_model, "judge_model"), + input_tokens=judge_account.input_tokens, + cache_hit_tokens=judge_account.cache_hit_tokens if judge_account.cache_split_reported else None, + cache_miss_tokens=judge_account.cache_miss_tokens if judge_account.cache_split_reported else None, + output_tokens=judge_account.output_tokens, + ), + "elapsed_ms": round((time.perf_counter_ns() - started_ns) / 1_000_000, 3), + }, + ) + if failed: + raise ScoreSmokeError(f"Scoring failed for {failed} question(s)") + return ScoreSmokeRun( + output_dir, manifest_path, inputs_path, results_path, judge_outputs_path, failures_path, summary_path + ) + + +def _judge_usage(usage: object) -> dict[str, object]: + """Keep the Judge's cache hit/miss split so its cost is priced per token class.""" + + if not isinstance(usage, Mapping): + return {"input_tokens": 0, "output_tokens": 0} + record: dict[str, object] = { + "input_tokens": _nonnegative_int(usage.get("input_tokens")), + "output_tokens": _nonnegative_int(usage.get("output_tokens")), + } + hit = _optional_nonnegative_int(usage.get("input_cache_hit_tokens")) + miss = _optional_nonnegative_int(usage.get("input_cache_miss_tokens")) + if hit is not None and miss is not None: + record["input_cache_hit_tokens"] = hit + record["input_cache_miss_tokens"] = miss + return record + + +def _judge_usage_totals( + *, + input_tokens: int, + output_tokens: int, + cache_hit_tokens: int, + cache_miss_tokens: int, + cache_split_reported: bool, +) -> dict[str, object]: + """Report summed Judge usage, keeping the cache split only when every call reported it.""" + + totals: dict[str, object] = {"input_tokens": input_tokens, "output_tokens": output_tokens} + if cache_split_reported: + totals["input_cache_hit_tokens"] = cache_hit_tokens + totals["input_cache_miss_tokens"] = cache_miss_tokens + return totals + + +def _score_one( + metrics: MetricsAPI, + question: dict[str, object], + reader: Mapping[str, object], + *, + sequence: int, + transport: ReaderTransport, +) -> tuple[dict[str, object], dict[str, object], dict[str, object] | None]: + question_id = _nonblank(question.get("id"), "question id") + response = _nonblank(reader.get("response_text"), "Reader response") + answer = _nonblank(question.get("answer"), "reference answer") + eval_spec = _nonblank(question.get("eval_function"), "eval function") + parsed = metrics.extract_boxed_answer(response) + evaluator = metrics.eval_name(eval_spec) + score_input = { + "schema": SCORE_INPUT_SCHEMA, + "question_id": question_id, + "eval_function": eval_spec, + "reference_answer": answer, + "reader_response": response, + "parsed_prediction": parsed, + "reader_response_sha256": hashlib.sha256(response.encode()).hexdigest(), + } + judge_output: dict[str, object] | None = None + if evaluator in LLM_JUDGE_NAMES: + messages = _judge_messages(metrics, evaluator, question, answer, response, parsed) + started_ns = time.perf_counter_ns() + try: + judge_response = transport.complete( + system=messages[0]["content"], + content=[{"type": "text", "text": messages[1]["content"]}], + ) + except ReaderResponseError as error: + # The Judge transport completed and reported usage; route it through the + # usage-preserving failure path instead of losing the accounting. + raise JudgeJudgementError( + f"Judge response for question {question_id} could not be read: {error}", + usage=error.usage, + judge_latency_ms=round((time.perf_counter_ns() - started_ns) / 1_000_000, 3), + ) from error + judge_latency_ms = round((time.perf_counter_ns() - started_ns) / 1_000_000, 3) + usage = _judge_usage(judge_response.get("usage")) + try: + judge_text = _response_text(judge_response) + label, reason = metrics._parse_llm_binary_judgement(judge_text) + except Exception as error: + detail = str(error).strip() or type(error).__name__ + raise JudgeJudgementError( + f"Judge judgement for question {question_id} could not be parsed: {detail[:200]}", + usage=usage, + judge_latency_ms=judge_latency_ms, + ) from error + judge_output = { + "schema": JUDGE_OUTPUT_SCHEMA, + "question_id": question_id, + "evaluator": evaluator, + "label": label, + "reason": reason, + "response_sha256": hashlib.sha256(judge_text.encode()).hexdigest(), + "usage": usage, + "judge_latency_ms": judge_latency_ms, + } + correct = label == 1 + mode = "llm_judge" + else: + correct = metrics.score_to_bool(metrics.eval_from_spec(eval_spec, parsed, answer)) + mode = "deterministic" + result = { + "schema": PER_QUESTION_SCHEMA, + "sequence": sequence, + "question_id": question_id, + "domain": question.get("domain"), + "eval_function": eval_spec, + "score_mode": mode, + "correct": correct, + "parsed_prediction": parsed, + "reader_response_sha256": hashlib.sha256(response.encode()).hexdigest(), + "judge_output_ref": None if judge_output is None else question_id, + } + return score_input, result, judge_output + + +def _judge_messages( + metrics: MetricsAPI, + evaluator: str, + question: dict[str, object], + answer: str, + response: str, + parsed: str, +) -> list[dict[str, str]]: + arguments = { + "question_text": _question_text(question), + "reference_answer": answer, + "model_full_response": response, + "model_final_answer": parsed, + } + if evaluator == "llm_abstention_checker": + return metrics._build_abstention_judge_messages(**arguments) + return metrics._build_gotchas_judge_messages(**arguments) + + +def _load_metrics(harness_root: Path) -> MetricsAPI: + root = str(harness_root.resolve()) + if root not in sys.path: + sys.path.insert(0, root) + return cast(MetricsAPI, importlib.import_module("evaluation.qa_eval_metrics")) + + +def _selected_questions(path: Path, selection: object) -> tuple[dict[str, object], ...]: + cases = getattr(selection, "cases", ()) + ids = [case.question_id for case in cases] + found: dict[str, dict[str, object]] = {} + for row in _jsonl(path, "question"): + question_id = row.get("id") + if isinstance(question_id, str) and question_id in ids: + found[question_id] = row + missing = [question_id for question_id in ids if question_id not in found] + if missing: + raise ScoreSmokeError(f"Missing smoke questions for scoring: {missing}") + return tuple(found[question_id] for question_id in ids) + + +def _reader_outputs(path: Path, selection: SmokeSelection) -> dict[str, dict[str, object]]: + expected = {case.question_id for case in selection.cases} + outputs: dict[str, dict[str, object]] = {} + for row in _jsonl(path, "reader output"): + question_id = row.get("question_id") + if not isinstance(question_id, str) or not question_id.strip(): + raise ScoreSmokeError("Reader output has an invalid question_id") + if question_id in outputs: + raise ScoreSmokeError(f"Reader outputs contain a duplicate question_id: {question_id}") + if question_id not in expected: + raise ScoreSmokeError(f"Reader output is not part of the fixed smoke subset: {question_id}") + outputs[question_id] = row + if set(outputs) != expected: + raise ScoreSmokeError("Reader outputs do not match the fixed smoke question ids") + return outputs + + +def _validate_reader_artifacts(manifest: dict[str, object], summary: dict[str, object]) -> None: + if manifest.get("classification") != "smoke-subset-reader-only": + raise ScoreSmokeError("Reader manifest is not a reader-only smoke subset") + if manifest.get("judge") is not None: + raise ScoreSmokeError("Reader artifacts must not contain Judge execution") + if summary.get("classification") != "smoke-subset-reader-only": + raise ScoreSmokeError("Reader summary is not a reader-only smoke subset") + if summary.get("question_count") != 10 or summary.get("failed") != 0: + raise ScoreSmokeError("Reader artifacts must contain ten successful smoke answers") + + +def _question_text(question: Mapping[str, object]) -> str: + value = question.get("question") + if isinstance(value, str): + return value + if isinstance(value, Mapping): + return _nonblank(value.get("text"), "question.text") + raise ScoreSmokeError("Question text is invalid") + + +def _response_text(response: Mapping[str, object]) -> str: + content = response.get("content") + if not isinstance(content, list): + raise ScoreSmokeError("Judge response content is invalid") + text_parts: list[str] = [] + for item in content: + if not isinstance(item, Mapping): + continue + text_part = item.get("text") + if isinstance(text_part, str): + text_parts.append(text_part) + text = "".join(text_parts).strip() + if not text: + raise ScoreSmokeError("Judge response contains no text") + return text + + +def _load_json(path: Path, label: str) -> dict[str, object]: + try: + value = json.loads(path.read_text(encoding="utf-8")) + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error: + raise ScoreSmokeError(f"Cannot read {label}: {path}") from error + if not isinstance(value, dict): + raise ScoreSmokeError(f"{label} must be a JSON object") + return value + + +def _jsonl(path: Path, label: str) -> Iterator[dict[str, object]]: + try: + with path.open(encoding="utf-8") as stream: + for line_number, line in enumerate(stream, start=1): + value = json.loads(line) + if not isinstance(value, dict): + raise ScoreSmokeError(f"{label} row {line_number} must be an object") + yield value + except (OSError, UnicodeDecodeError, json.JSONDecodeError) as error: + raise ScoreSmokeError(f"Cannot read {label} input: {path}") from error + + +def _file_digest(path: Path) -> str: + hasher = hashlib.sha256() + try: + with path.open("rb") as stream: + while chunk := stream.read(1024 * 1024): + hasher.update(chunk) + except OSError as error: + raise ScoreSmokeError(f"Cannot hash score input: {path}") from error + return hasher.hexdigest() + + +def _write_json_exclusive(path: Path, value: object) -> None: + _write_text_exclusive(path, json.dumps(value, ensure_ascii=True, indent=2, sort_keys=True) + "\n") + + +def _write_text_exclusive(path: Path, value: str) -> None: + try: + with path.open("x", encoding="utf-8", newline="") as stream: + stream.write(value) + except OSError as error: + raise ScoreSmokeError(f"Cannot write score artifact: {path}") from error + + +def _create_empty(path: Path) -> None: + _write_text_exclusive(path, "") + + +def _append_json(path: Path, value: object) -> None: + with path.open("a", encoding="utf-8", newline="") as stream: + stream.write(json.dumps(value, ensure_ascii=False, separators=(",", ":"), allow_nan=False) + "\n") + + +def _nonblank(value: object, label: str) -> str: + if not isinstance(value, str) or not value.strip(): + raise ScoreSmokeError(f"{label} must be a non-empty string") + return value.strip() + + +def _nonnegative_int(value: object) -> int: + return value if isinstance(value, int) and not isinstance(value, bool) and value >= 0 else 0 + + +def _optional_nonnegative_int(value: object) -> int | None: + """Keep an absent cache field distinguishable from a reported zero.""" + + if isinstance(value, int) and not isinstance(value, bool) and value >= 0: + return value + return None diff --git a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/smoke.py b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/smoke.py index 8f5f66dfc1..e3601e3c96 100644 --- a/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/smoke.py +++ b/evaluation/src/powercontext_eval/benchmarks/longmemeval_v2/smoke.py @@ -26,7 +26,8 @@ UPSTREAM_HARNESS_COMMIT, UPSTREAM_REPOSITORY, LongMemEvalV2Catalog, - LongMemEvalV2CatalogError, + LongMemEvalV2EnvironmentError, + LongMemEvalV2InputError, load_dataset_lock, load_smoke_manifest, validate_harness_checkout, @@ -55,18 +56,22 @@ def prepare_smoke_run( lock = load_dataset_lock(dataset_lock) selection = load_smoke_manifest(smoke_manifest) if selection.tier != lock.tier: - raise LongMemEvalV2CatalogError("Smoke manifest tier does not match the dataset lock") + raise LongMemEvalV2InputError("Smoke manifest tier does not match the dataset lock") validate_harness_checkout(harness_root) catalog = LongMemEvalV2Catalog.load(data_root, tier=selection.tier, expected_digests=lock.file_digests) - catalog.select_smoke(selection.cases) + validated = catalog.select_smoke(selection.cases) + lock_content_digest = _content_digest(dataset_lock, "dataset lock") + smoke_content_digest = _content_digest(smoke_manifest, "smoke manifest") try: output_dir.mkdir(parents=True, exist_ok=False) except FileExistsError as error: - raise LongMemEvalV2CatalogError(f"Refusing to overwrite smoke artifacts: {output_dir}") from error + raise LongMemEvalV2EnvironmentError(f"Refusing to overwrite smoke artifacts: {output_dir}") from error + except OSError as error: + raise LongMemEvalV2EnvironmentError(f"Cannot create smoke artifact directory: {output_dir}") from error subset_path = output_dir / "subset.json" manifest_path = output_dir / "manifest.json" - subset = selection.as_json() + subset = validated.as_run_artifact_json() _write_json(subset_path, subset) _write_json( manifest_path, @@ -82,10 +87,10 @@ def prepare_smoke_run( "revision": lock.dataset_revision, "tier": catalog.tier, }, - "dataset_lock": {"content_sha256": hashlib.sha256(dataset_lock.read_bytes()).hexdigest()}, + "dataset_lock": {"content_sha256": lock_content_digest}, "smoke_manifest": { - "content_sha256": hashlib.sha256(smoke_manifest.read_bytes()).hexdigest(), - "subset_sha256": hashlib.sha256(subset_path.read_bytes()).hexdigest(), + "content_sha256": smoke_content_digest, + "subset_sha256": _content_digest(subset_path, "subset artifact"), }, }, ) @@ -93,4 +98,14 @@ def prepare_smoke_run( def _write_json(path: Path, value: object) -> None: - path.write_text(json.dumps(value, ensure_ascii=True, indent=2, sort_keys=True) + "\n", encoding="utf-8") + try: + path.write_text(json.dumps(value, ensure_ascii=True, indent=2, sort_keys=True) + "\n", encoding="utf-8") + except OSError as error: + raise LongMemEvalV2EnvironmentError(f"Cannot write smoke artifact: {path}") from error + + +def _content_digest(path: Path, label: str) -> str: + try: + return hashlib.sha256(path.read_bytes()).hexdigest() + except OSError as error: + raise LongMemEvalV2EnvironmentError(f"Cannot read LongMemEval-V2 {label}: {path}") from error diff --git a/evaluation/src/powercontext_eval/cli.py b/evaluation/src/powercontext_eval/cli.py index 6d06224f92..bba3ff4082 100644 --- a/evaluation/src/powercontext_eval/cli.py +++ b/evaluation/src/powercontext_eval/cli.py @@ -32,14 +32,49 @@ import typer from pydantic import ValidationError -from powercontext_eval.benchmarks.longmemeval_v2.catalog import LongMemEvalV2CatalogError +from powercontext_eval.benchmarks.longmemeval_v2.adapter import PowerContextMemoryAdapterError +from powercontext_eval.benchmarks.longmemeval_v2.arms import DEFAULT_EXPERIMENT_ARM_ID, ExperimentArmError +from powercontext_eval.benchmarks.longmemeval_v2.catalog import ( + LongMemEvalV2CatalogError, + LongMemEvalV2EnvironmentError, + LongMemEvalV2InputError, +) +from powercontext_eval.benchmarks.longmemeval_v2.costs import ( + CostPolicyError, + ModelPricePolicy, + parse_cost_policy, +) +from powercontext_eval.benchmarks.longmemeval_v2.prepare_smoke import ( + DEFAULT_PROCESSOR_MODEL, + PrepareSmokeError, + prepare_reader_inputs_smoke, +) +from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import ( + DEFAULT_ANTHROPIC_BASE_URL_ENV, + ReaderSmokeError, + run_reader_smoke, +) +from powercontext_eval.benchmarks.longmemeval_v2.replay_score import ReplayScoreError, replay_score_smoke +from powercontext_eval.benchmarks.longmemeval_v2.report import ReportError, build_report +from powercontext_eval.benchmarks.longmemeval_v2.retrieval_smoke import RetrievalSmokeError, run_retrieval_smoke +from powercontext_eval.benchmarks.longmemeval_v2.run_smoke import RunSmokeError, run_smoke +from powercontext_eval.benchmarks.longmemeval_v2.score_smoke import ( + DEFAULT_DEEPSEEK_BASE_URL as SCORE_DEFAULT_DEEPSEEK_BASE_URL, +) +from powercontext_eval.benchmarks.longmemeval_v2.score_smoke import ( + DEFAULT_DEEPSEEK_MODEL as SCORE_DEFAULT_DEEPSEEK_MODEL, +) +from powercontext_eval.benchmarks.longmemeval_v2.score_smoke import ( + DEFAULT_DEEPSEEK_TOKEN_ENV as SCORE_DEFAULT_DEEPSEEK_TOKEN_ENV, +) +from powercontext_eval.benchmarks.longmemeval_v2.score_smoke import ( + ScoreSmokeError, + run_score_smoke, +) from powercontext_eval.benchmarks.longmemeval_v2.smoke import prepare_smoke_run from powercontext_eval.benchmarks.swebench_pro.catalog import PUBLIC_V2_TASK_SET, SweBenchProCatalog, TaskSet from powercontext_eval.codex import DEFAULT_CODEX_MODEL, DEFAULT_REASONING_EFFORT from powercontext_eval.models import TreatmentMode -from powercontext_eval.powercontext_sut import DEFAULT_DOCKER_NETWORK_POOL, run_codex_contract_smoke -from powercontext_eval.runner import RunConfig, run_swebench_pro_instance -from powercontext_eval.web.batches import BatchCreate if TYPE_CHECKING: from powercontext_eval.web.config import WebConfig @@ -49,6 +84,7 @@ longmemeval_v2_app = typer.Typer(no_args_is_help=True, help="Pinned LongMemEval-V2 evaluation.") app.add_typer(swebench_pro_app, name="swebench-pro") app.add_typer(longmemeval_v2_app, name="longmemeval-v2") +DEFAULT_DOCKER_NETWORK_POOL = "172.30.0.0/15" @app.callback() @@ -60,6 +96,37 @@ class _Stoppable(Protocol): def stop(self) -> None: ... +def run_codex_contract_smoke(**kwargs: Any) -> Any: + """Load the platform-specific contract runner only when its command is used.""" + + from powercontext_eval.powercontext_sut import run_codex_contract_smoke as implementation + + return implementation(**kwargs) + + +def run_swebench_pro_instance(*args: Any, **kwargs: Any) -> Any: + """Load the platform-specific SWE-bench runner only when its command is used.""" + + from powercontext_eval.runner import run_swebench_pro_instance as implementation + + return implementation(*args, **kwargs) + + +def _parse_price_policy_option(value: str | None, *, label: str) -> ModelPricePolicy | None: + """Parse an explicitly passed JSON price policy, or keep costs unconfigured as null.""" + + if value is None: + return None + try: + parsed = json.loads(value) + except json.JSONDecodeError as error: + raise typer.BadParameter(f"{label} must be a JSON object: {error}") from None + try: + return parse_cost_policy(parsed, label=label) + except CostPolicyError as error: + raise typer.BadParameter(str(error)) from None + + def _request_worker_stop(worker: _Stoppable, _signum: int, _frame: FrameType | None) -> None: """Request that a worker exit after its current task finishes.""" worker.stop() @@ -175,11 +242,11 @@ def codex_contract_smoke( @longmemeval_v2_app.command("smoke") def longmemeval_v2_smoke( - data_root: Path = typer.Option(..., "--data-root"), - dataset_lock: Path = typer.Option(..., "--dataset-lock"), - harness_root: Path = typer.Option(..., "--harness-root"), - smoke_manifest: Path = typer.Option(..., "--smoke-manifest"), - output_dir: Path = typer.Option(..., "--output-dir"), + data_root: Annotated[Path, typer.Option("--data-root")], + dataset_lock: Annotated[Path, typer.Option("--dataset-lock")], + harness_root: Annotated[Path, typer.Option("--harness-root")], + smoke_manifest: Annotated[Path, typer.Option("--smoke-manifest")], + output_dir: Annotated[Path, typer.Option("--output-dir")], ) -> None: """Validate fixed LongMemEval-V2 inputs and write smoke artifacts without calling a model.""" @@ -191,8 +258,11 @@ def longmemeval_v2_smoke( smoke_manifest=smoke_manifest, output_dir=output_dir, ) - except LongMemEvalV2CatalogError as error: + except LongMemEvalV2InputError as error: raise typer.BadParameter(str(error)) from None + except LongMemEvalV2EnvironmentError as error: + typer.echo(f"LongMemEval-V2 smoke failed: {error}", err=True) + raise typer.Exit(code=1) from None typer.echo( json.dumps( { @@ -206,6 +276,356 @@ def longmemeval_v2_smoke( ) +@longmemeval_v2_app.command("retrieval-smoke") +def longmemeval_v2_retrieval_smoke( + data_root: Annotated[Path, typer.Option("--data-root")], + dataset_lock: Annotated[Path, typer.Option("--dataset-lock")], + harness_root: Annotated[Path, typer.Option("--harness-root")], + smoke_manifest: Annotated[Path, typer.Option("--smoke-manifest")], + output_dir: Annotated[Path, typer.Option("--output-dir")], + run_id: Annotated[str, typer.Option("--run-id")], + powercontext_revision: Annotated[str, typer.Option("--powercontext-revision")], + integration_revision: Annotated[str, typer.Option("--integration-revision")], + base_url: Annotated[str, typer.Option("--base-url")] = "http://127.0.0.1:8000", + token_env: Annotated[str, typer.Option("--token-env")] = "POWERCONTEXT_TOKEN", + experiment_arm: Annotated[str, typer.Option("--experiment-arm")] = DEFAULT_EXPERIMENT_ARM_ID, + search_limit: Annotated[int, typer.Option("--search-limit", min=1, max=50)] = 10, + timeout_seconds: Annotated[float, typer.Option("--timeout-seconds", min=0.1)] = 30.0, +) -> None: + """Run the fixed LongMemEval-V2 subset through Memory retrieval without a model.""" + + try: + result = run_retrieval_smoke( + data_root=data_root, + dataset_lock=dataset_lock, + harness_root=harness_root, + smoke_manifest=smoke_manifest, + output_dir=output_dir, + run_id=run_id, + powercontext_revision=powercontext_revision, + integration_revision=integration_revision, + base_url=base_url, + token_env=token_env, + experiment_arm=experiment_arm, + search_limit=search_limit, + timeout_seconds=timeout_seconds, + ) + except ExperimentArmError as error: + raise typer.BadParameter(str(error)) from None + except (LongMemEvalV2CatalogError, RetrievalSmokeError, PowerContextMemoryAdapterError) as error: + typer.echo(f"LongMemEval-V2 retrieval smoke failed: {error}", err=True) + raise typer.Exit(code=1) from None + typer.echo( + json.dumps( + { + "classification": "smoke-subset-retrieval-only", + "manifest": str(result.manifest_path), + "results": str(result.results_path), + "summary": str(result.summary_path), + }, + ensure_ascii=False, + sort_keys=True, + ) + ) + + +@longmemeval_v2_app.command("prepare-smoke") +def longmemeval_v2_prepare_smoke( + retrieval_dir: Annotated[Path, typer.Option("--retrieval-dir")], + harness_root: Annotated[Path, typer.Option("--harness-root")], + harness_python: Annotated[Path, typer.Option("--harness-python")], + output_dir: Annotated[Path, typer.Option("--output-dir")], + processor_revision: Annotated[str, typer.Option("--processor-revision")], + processor_model: Annotated[str, typer.Option("--processor-model")] = DEFAULT_PROCESSOR_MODEL, + memory_context_max_tokens: Annotated[int, typer.Option("--memory-context-max-tokens", min=1)] = 200_000, +) -> None: + """Build pinned, bounded Reader inputs from retrieval-only smoke artifacts.""" + + try: + result = prepare_reader_inputs_smoke( + retrieval_dir=retrieval_dir, + harness_root=harness_root, + harness_python=harness_python, + output_dir=output_dir, + processor_model=processor_model, + processor_revision=processor_revision, + memory_context_max_tokens=memory_context_max_tokens, + ) + except PrepareSmokeError as error: + typer.echo(f"LongMemEval-V2 prompt preparation failed: {error}", err=True) + raise typer.Exit(code=1) from None + typer.echo( + json.dumps( + { + "classification": "smoke-subset-prepare-only", + "manifest": str(result.manifest_path), + "prompts": str(result.prompts_path), + "summary": str(result.summary_path), + }, + ensure_ascii=False, + sort_keys=True, + ) + ) + + +@longmemeval_v2_app.command("reader-smoke") +def longmemeval_v2_reader_smoke( + prepared_dir: Annotated[Path, typer.Option("--prepared-dir")], + output_dir: Annotated[Path, typer.Option("--output-dir")], + provider: Annotated[str, typer.Option("--provider")] = "anthropic-compatible", + model: Annotated[str | None, typer.Option("--model")] = None, + base_url: Annotated[str | None, typer.Option("--base-url")] = None, + base_url_env: Annotated[str, typer.Option("--base-url-env")] = DEFAULT_ANTHROPIC_BASE_URL_ENV, + token_env: Annotated[str | None, typer.Option("--token-env")] = None, + max_tokens: Annotated[int, typer.Option("--max-tokens", min=1)] = 512, + temperature: Annotated[float, typer.Option("--temperature", min=0.0, max=2.0)] = 0.0, + timeout_seconds: Annotated[float, typer.Option("--timeout-seconds", min=1.0)] = 120.0, + max_questions: Annotated[int | None, typer.Option("--max-questions", min=1)] = None, + price_policy: Annotated[str | None, typer.Option("--price-policy")] = None, +) -> None: + """Call a configured Reader over prepared smoke prompts without scoring.""" + + resolved_price_policy = _parse_price_policy_option(price_policy, label="--price-policy") + try: + result = run_reader_smoke( + prepared_dir=prepared_dir, + output_dir=output_dir, + provider=provider, + model=model, + base_url=base_url, + base_url_env=base_url_env, + token_env=token_env, + max_tokens=max_tokens, + temperature=temperature, + timeout_seconds=timeout_seconds, + max_questions=max_questions, + price_policy=resolved_price_policy, + ) + except ReaderSmokeError as error: + typer.echo(f"LongMemEval-V2 Reader failed: {error}", err=True) + raise typer.Exit(code=1) from None + typer.echo( + json.dumps( + { + "classification": "smoke-subset-reader-only", + "manifest": str(result.manifest_path), + "outputs": str(result.outputs_path), + "summary": str(result.summary_path), + }, + ensure_ascii=False, + sort_keys=True, + ) + ) + + +@longmemeval_v2_app.command("score-smoke") +def longmemeval_v2_score_smoke( + reader_dir: Annotated[Path, typer.Option("--reader-dir")], + data_root: Annotated[Path, typer.Option("--data-root")], + dataset_lock: Annotated[Path, typer.Option("--dataset-lock")], + smoke_manifest: Annotated[Path, typer.Option("--smoke-manifest")], + harness_root: Annotated[Path, typer.Option("--harness-root")], + output_dir: Annotated[Path, typer.Option("--output-dir")], + judge_model: Annotated[str, typer.Option("--judge-model")] = SCORE_DEFAULT_DEEPSEEK_MODEL, + judge_token_env: Annotated[str, typer.Option("--judge-token-env")] = SCORE_DEFAULT_DEEPSEEK_TOKEN_ENV, + judge_base_url: Annotated[str, typer.Option("--judge-base-url")] = SCORE_DEFAULT_DEEPSEEK_BASE_URL, + judge_max_tokens: Annotated[int, typer.Option("--judge-max-tokens", min=1)] = 256, + judge_temperature: Annotated[float, typer.Option("--judge-temperature", min=0.0, max=2.0)] = 0.0, + judge_timeout_seconds: Annotated[float, typer.Option("--judge-timeout-seconds", min=1.0)] = 120.0, + price_policy: Annotated[str | None, typer.Option("--judge-price-policy")] = None, +) -> None: + """Score Reader smoke outputs with pinned rules and DeepSeek only where upstream requires a judge.""" + + resolved_price_policy = _parse_price_policy_option(price_policy, label="--judge-price-policy") + try: + result = run_score_smoke( + reader_dir=reader_dir, + data_root=data_root, + dataset_lock=dataset_lock, + smoke_manifest=smoke_manifest, + harness_root=harness_root, + output_dir=output_dir, + judge_model=judge_model, + judge_token_env=judge_token_env, + judge_base_url=judge_base_url, + judge_max_tokens=judge_max_tokens, + judge_temperature=judge_temperature, + judge_timeout_seconds=judge_timeout_seconds, + judge_price_policy=resolved_price_policy, + ) + except ScoreSmokeError as error: + typer.echo(f"LongMemEval-V2 scoring failed: {error}", err=True) + raise typer.Exit(code=1) from None + typer.echo( + json.dumps( + { + "classification": "smoke-subset-score-only", + "manifest": str(result.manifest_path), + "results": str(result.results_path), + "summary": str(result.summary_path), + }, + ensure_ascii=False, + sort_keys=True, + ) + ) + + +@longmemeval_v2_app.command("replay-score") +def longmemeval_v2_replay_score( + score_dir: Annotated[Path, typer.Option("--score-dir")], + harness_root: Annotated[Path, typer.Option("--harness-root")], + output_dir: Annotated[Path, typer.Option("--output-dir")], +) -> None: + """Replay saved deterministic and Judge score decisions without model calls.""" + + try: + result = replay_score_smoke(score_dir=score_dir, harness_root=harness_root, output_dir=output_dir) + except ReplayScoreError as error: + typer.echo(f"LongMemEval-V2 score replay failed: {error}", err=True) + raise typer.Exit(code=1) from None + typer.echo( + json.dumps( + { + "classification": "smoke-subset-score-replay", + "manifest": str(result.manifest_path), + "results": str(result.results_path), + "summary": str(result.summary_path), + }, + ensure_ascii=False, + sort_keys=True, + ) + ) + + +@longmemeval_v2_app.command("run-smoke") +def longmemeval_v2_run_smoke( + data_root: Annotated[Path, typer.Option("--data-root")], + dataset_lock: Annotated[Path, typer.Option("--dataset-lock")], + smoke_manifest: Annotated[Path, typer.Option("--smoke-manifest")], + harness_root: Annotated[Path, typer.Option("--harness-root")], + harness_python: Annotated[Path, typer.Option("--harness-python")], + processor_revision: Annotated[str, typer.Option("--processor-revision")], + output_dir: Annotated[Path, typer.Option("--output-dir")], + powercontext_revision: Annotated[str, typer.Option("--powercontext-revision")], + integration_revision: Annotated[str, typer.Option("--integration-revision")], + run_id: Annotated[str | None, typer.Option("--run-id")] = None, + processor_model: Annotated[str, typer.Option("--processor-model")] = DEFAULT_PROCESSOR_MODEL, + memory_context_max_tokens: Annotated[int, typer.Option("--memory-context-max-tokens", min=1)] = 200_000, + powercontext_base_url: Annotated[str, typer.Option("--powercontext-base-url")] = "http://127.0.0.1:8000", + powercontext_token_env: Annotated[str, typer.Option("--powercontext-token-env")] = "POWERCONTEXT_TOKEN", + experiment_arm: Annotated[str, typer.Option("--experiment-arm")] = DEFAULT_EXPERIMENT_ARM_ID, + search_limit: Annotated[int, typer.Option("--search-limit", min=1, max=50)] = 10, + timeout_seconds: Annotated[float, typer.Option("--timeout-seconds", min=0.1)] = 30.0, + reader_provider: Annotated[str, typer.Option("--reader-provider")] = "deepseek-openai", + reader_model: Annotated[str | None, typer.Option("--reader-model")] = None, + reader_base_url: Annotated[str | None, typer.Option("--reader-base-url")] = None, + reader_base_url_env: Annotated[str, typer.Option("--reader-base-url-env")] = DEFAULT_ANTHROPIC_BASE_URL_ENV, + reader_token_env: Annotated[str | None, typer.Option("--reader-token-env")] = None, + reader_max_tokens: Annotated[int, typer.Option("--reader-max-tokens", min=1)] = 512, + reader_temperature: Annotated[float, typer.Option("--reader-temperature", min=0.0, max=2.0)] = 0.0, + reader_timeout_seconds: Annotated[float, typer.Option("--reader-timeout-seconds", min=1.0)] = 120.0, + judge_provider: Annotated[str, typer.Option("--judge-provider")] = "deepseek-openai", + judge_model: Annotated[str, typer.Option("--judge-model")] = SCORE_DEFAULT_DEEPSEEK_MODEL, + judge_token_env: Annotated[str, typer.Option("--judge-token-env")] = SCORE_DEFAULT_DEEPSEEK_TOKEN_ENV, + judge_base_url: Annotated[str, typer.Option("--judge-base-url")] = SCORE_DEFAULT_DEEPSEEK_BASE_URL, + judge_max_tokens: Annotated[int, typer.Option("--judge-max-tokens", min=1)] = 256, + judge_temperature: Annotated[float, typer.Option("--judge-temperature", min=0.0, max=2.0)] = 0.0, + judge_timeout_seconds: Annotated[float, typer.Option("--judge-timeout-seconds", min=1.0)] = 120.0, + price_policy: Annotated[str | None, typer.Option("--price-policy")] = None, + judge_price_policy: Annotated[str | None, typer.Option("--judge-price-policy")] = None, + skip_reader: Annotated[bool, typer.Option("--skip-reader")] = False, + skip_score: Annotated[bool, typer.Option("--skip-score")] = False, +) -> None: + """Run the whole LongMemEval-V2 smoke workload into one fail-closed run directory.""" + + resolved_price_policy = _parse_price_policy_option(price_policy, label="--price-policy") + resolved_judge_price_policy = _parse_price_policy_option(judge_price_policy, label="--judge-price-policy") + try: + result = run_smoke( + data_root=data_root, + dataset_lock=dataset_lock, + smoke_manifest=smoke_manifest, + harness_root=harness_root, + harness_python=harness_python, + processor_revision=processor_revision, + output_dir=output_dir, + powercontext_revision=powercontext_revision, + integration_revision=integration_revision, + run_id=run_id, + processor_model=processor_model, + memory_context_max_tokens=memory_context_max_tokens, + powercontext_base_url=powercontext_base_url, + powercontext_token_env=powercontext_token_env, + experiment_arm=experiment_arm, + search_limit=search_limit, + timeout_seconds=timeout_seconds, + reader_provider=reader_provider, + reader_model=reader_model, + reader_base_url=reader_base_url, + reader_base_url_env=reader_base_url_env, + reader_token_env=reader_token_env, + reader_max_tokens=reader_max_tokens, + reader_temperature=reader_temperature, + reader_timeout_seconds=reader_timeout_seconds, + judge_provider=judge_provider, + judge_model=judge_model, + judge_token_env=judge_token_env, + judge_base_url=judge_base_url, + judge_max_tokens=judge_max_tokens, + judge_temperature=judge_temperature, + judge_timeout_seconds=judge_timeout_seconds, + reader_price_policy=resolved_price_policy, + judge_price_policy=resolved_judge_price_policy, + skip_reader=skip_reader, + skip_score=skip_score, + ) + except (RunSmokeError, LongMemEvalV2CatalogError, ExperimentArmError) as error: + typer.echo(f"LongMemEval-V2 smoke run failed: {error}", err=True) + raise typer.Exit(code=1) from None + typer.echo( + json.dumps( + { + "classification": "smoke-subset", + "status": result.status, + "manifest": str(result.manifest_path), + "summary": str(result.summary_path), + "completed_phases": list(result.completed_phases), + "skipped_phases": list(result.skipped_phases), + "failed_phase": result.failed_phase, + }, + ensure_ascii=False, + sort_keys=True, + ) + ) + if result.status != "completed": + raise typer.Exit(code=1) + + +@longmemeval_v2_app.command("report") +def longmemeval_v2_report( + run_dir: Annotated[Path, typer.Option("--run-dir")], + output_dir: Annotated[Path | None, typer.Option("--output-dir")] = None, +) -> None: + """Summarize one saved smoke run into a single unified report without a model or server.""" + + try: + result = build_report(run_dir=run_dir, output_dir=output_dir) + except ReportError as error: + typer.echo(f"LongMemEval-V2 report failed: {error}", err=True) + raise typer.Exit(code=1) from None + typer.echo( + json.dumps( + { + "classification": "smoke-subset", + "report": str(result.report_path), + "markdown": str(result.markdown_path), + }, + ensure_ascii=False, + sort_keys=True, + ) + ) + + @swebench_pro_app.command("run") def swebench_pro_run( root_path: str = typer.Option(..., "--root"), @@ -232,6 +652,8 @@ def swebench_pro_run( ) -> None: """Run Gold, PowerContext OFF/ON, official grading, and report generation.""" + from powercontext_eval.runner import RunConfig + root = Path(root_path) harness = Path(harness_root) if harness_root is not None else root / "cache" / "swebench-pro.git" dataset = Path(dataset_path) if dataset_path is not None else harness / "helper_code" / "sweap_eval_full_v2.jsonl" @@ -296,12 +718,14 @@ def swebench_pro_create_batch( powercontext_ref: str = typer.Option("latest", "--powercontext-ref"), task_set: str = typer.Option(PUBLIC_V2_TASK_SET, "--task-set"), model: str = typer.Option(DEFAULT_CODEX_MODEL, "--model"), - treatment_mode: TreatmentMode = typer.Option(TreatmentMode.OFF_ON, "--treatment-mode"), + treatment_mode: Annotated[TreatmentMode, typer.Option("--treatment-mode")] = TreatmentMode.OFF_ON, usage_pause_percent: int = typer.Option(80, "--usage-pause-percent", min=1, max=100), start_paused: bool = typer.Option(False, "--start-paused/--start-running"), ) -> None: """Create one full batch through the console API, optionally atomically paused.""" + from powercontext_eval.web.batches import BatchCreate + endpoint = _batch_api_endpoint(console_url) try: batch = BatchCreate( diff --git a/evaluation/tests/unit/test_cli.py b/evaluation/tests/unit/test_cli.py index 5450a0642f..458b009bb1 100644 --- a/evaluation/tests/unit/test_cli.py +++ b/evaluation/tests/unit/test_cli.py @@ -24,8 +24,12 @@ from typer.testing import CliRunner -from powercontext_eval.cli import app +from powercontext_eval.benchmarks.longmemeval_v2.catalog import ( + LongMemEvalV2EnvironmentError, + LongMemEvalV2InputError, +) from powercontext_eval.benchmarks.longmemeval_v2.smoke import PreparedSmokeRun +from powercontext_eval.cli import app from powercontext_eval.runner import MinimalRunResult, RunConfig @@ -75,6 +79,46 @@ def prepare(**kwargs: object) -> PreparedSmokeRun: assert '"classification": "smoke-subset"' in result.output +def test_longmemeval_v2_smoke_reports_invalid_configuration_as_bad_parameter(monkeypatch) -> None: + def prepare(**_kwargs: object) -> PreparedSmokeRun: + raise LongMemEvalV2InputError("Smoke manifest schema is unsupported") + + monkeypatch.setattr("powercontext_eval.cli.prepare_smoke_run", prepare) + result = CliRunner().invoke(app, ["longmemeval-v2", "smoke", *_longmemeval_smoke_arguments()]) + + assert result.exit_code == 2 + assert "Invalid value" in result.output + assert "Smoke manifest schema is unsupported" in result.output + + +def test_longmemeval_v2_smoke_reports_environment_failure_with_exit_one(monkeypatch) -> None: + def prepare(**_kwargs: object) -> PreparedSmokeRun: + raise LongMemEvalV2EnvironmentError("LongMemEval-V2 SHA-256 mismatch for questions.jsonl") + + monkeypatch.setattr("powercontext_eval.cli.prepare_smoke_run", prepare) + result = CliRunner().invoke(app, ["longmemeval-v2", "smoke", *_longmemeval_smoke_arguments()]) + + assert result.exit_code == 1 + assert "LongMemEval-V2 smoke failed" in result.output + assert "SHA-256 mismatch" in result.output + assert "Invalid value" not in result.output + + +def _longmemeval_smoke_arguments() -> list[str]: + return [ + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--harness-root", + "/harness", + "--smoke-manifest", + "/smoke.json", + "--output-dir", + "/output", + ] + + def test_codex_contract_smoke_is_an_executable_injectable_cli(monkeypatch) -> None: calls: list[dict[str, object]] = [] diff --git a/evaluation/tests/unit/test_longmemeval_v2_adapter.py b/evaluation/tests/unit/test_longmemeval_v2_adapter.py new file mode 100644 index 0000000000..4197cb53a5 --- /dev/null +++ b/evaluation/tests/unit/test_longmemeval_v2_adapter.py @@ -0,0 +1,473 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from __future__ import annotations + +import json +from collections.abc import Mapping +from pathlib import Path +from typing import Any + +import pytest + +from powercontext_eval.benchmarks.longmemeval_v2.adapter import ( + AUDIT_SCHEMA, + PowerContextHTTPRuntime, + PowerContextMemory, + PowerContextMemoryAdapterError, + PowerContextMemoryModeError, +) + + +class FakeRuntime: + def __init__(self) -> None: + self.captures: list[dict[str, object]] = [] + self.memories: list[dict[str, object]] = [] + self.searches: list[dict[str, object]] = [] + + def capture_content_source(self, payload: Mapping[str, object]) -> Mapping[str, object]: + request = dict(payload) + self.captures.append(request) + return { + "status": "accepted", + "source": {"name": "content", "source_id": request["source_id"]}, + "position": len(self.captures), + } + + def remember_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + request = dict(payload) + self.memories.append(request) + index = len(self.memories) + return { + "memory": {"family": "memory", "artifact_id": "memory", "revision": index}, + "entry": { + "citation": { + "memory_ref": {"family": "memory", "artifact_id": "memory", "revision": index}, + "entry_id": f"entry-{index}", + "entry_version_id": f"entry-{index}-v1", + } + }, + } + + def search_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + self.searches.append(dict(payload)) + return { + "mode": payload.get("mode"), + "hits": [ + { + "text": "Use the Network assignment group.", + "score": 0.9, + "matched_by": ["fts"], + "citation": { + "memory_ref": {"family": "memory", "artifact_id": "memory", "revision": 2}, + "entry_id": "entry-2", + "entry_version_id": "entry-2-v1", + }, + } + ], + } + + def prepare_context(self, payload: Mapping[str, object]) -> Mapping[str, object]: + content = "Prepared compact context." + return { + "schema": "powercontext.prepared-context.v1", + "status": "ready", + "content": content, + "content_bytes": len(content.encode()), + } + + +class GuardedTrajectory(dict[str, object]): + def get(self, key: object, default: object = None) -> object: + if key in {"question_type", "answer", "gold_answer", "judge", "scorer"}: + raise AssertionError(f"adapter accessed forbidden field {key}") + return super().get(key, default) + + +def adapter(tmp_path: Path, runtime: FakeRuntime, **params: object) -> PowerContextMemory: + memory = PowerContextMemory( + { + "scope_id": "benchmark-run", + "audit_path": str(tmp_path / "adapter-audit.jsonl"), + **params, + } + ) + memory.configure_runtime(runtime=runtime) + return memory + + +def trajectory() -> GuardedTrajectory: + return GuardedTrajectory( + { + "id": "trajectory-1", + "domain": "enterprise", + "environment": "workarena", + "goal": "Assign an incident.", + "outcome": "success", + "start_url": "https://example.test/start", + "states": [ + { + "state_index": 0, + "step": 0, + "url": "https://example.test/start", + "action": None, + "thought": "Find the incident.", + "accessibility_tree": "Incident list " + "记忆" * 500, + "screenshot": "screenshots/trajectory-1/0.png", + } + ], + "question_type": "must not be read", + "gold_answer": "must not be read", + "judge": {"must": "not be read"}, + } + ) + + +def audit_events(tmp_path: Path) -> list[dict[str, Any]]: + return [json.loads(line) for line in (tmp_path / "adapter-audit.jsonl").read_text(encoding="utf-8").splitlines()] + + +def test_insert_uses_public_source_and_memory_operations_with_utf8_safe_chunks(tmp_path: Path) -> None: + runtime = FakeRuntime() + memory = adapter(tmp_path, runtime, source_chunk_bytes=512) + + memory.insert(trajectory()) + + assert len(runtime.captures) > 1 + assert len(runtime.memories) == 1 + assert all(request["scope_id"] == "benchmark-run" for request in runtime.captures) + assert all(len(str(request["content"]).encode()) <= 512 for request in runtime.captures) + restored = "".join(str(request["content"]) for request in runtime.captures) + projected = json.loads(restored) + assert set(projected) == {"id", "domain", "environment", "goal", "outcome", "start_url", "states"} + assert "question_type" not in restored + assert "gold_answer" not in restored + assert len(str(runtime.memories[0]["text"]).encode()) <= 8192 + assert "Assign an incident" in str(runtime.memories[0]["text"]) + + [event] = audit_events(tmp_path) + assert event["schema"] == AUDIT_SCHEMA + assert event["operation"] == "ingest" + assert event["status"] == "succeeded" + assert event["scope_id"] == "benchmark-run" + assert isinstance(event["observed_at"], str) + assert event["source_chunk_count"] == len(runtime.captures) + assert event["memory_entry_count"] == 1 + assert len(event["source_refs"]) == len(runtime.captures) + assert len(event["memory_citations"]) == 1 + assert set(event["timings_ms"]) == {"source_capture", "memory_remember", "total"} + + +def test_l0_l1_projection_writes_two_bounded_memory_entries_without_schema_changes(tmp_path: Path) -> None: + runtime = FakeRuntime() + memory = adapter(tmp_path, runtime, memory_projection="deterministic-l0-l1-v1") + + memory.insert(trajectory()) + + assert len(runtime.memories) == 2 + assert str(runtime.memories[0]["text"]).startswith("LongMemEval-V2 deterministic L0 trajectory index") + assert str(runtime.memories[1]["text"]).startswith("LongMemEval-V2 deterministic L1 trajectory summary") + assert all(len(str(entry["text"]).encode()) <= 8_192 for entry in runtime.memories) + [audit] = audit_events(tmp_path) + assert audit["memory_projection"] == "deterministic-l0-l1-v1" + assert audit["memory_entry_count"] == 2 + + +def test_l0_l1_projection_keeps_the_byte_limit_when_a_trajectory_reaches_it(tmp_path: Path) -> None: + runtime = FakeRuntime() + memory = adapter(tmp_path, runtime, memory_projection="deterministic-l0-l1-v1") + long_trajectory = trajectory() + long_trajectory["states"] = [ + { + "state_index": index, + "step": index, + "url": "https://example.test/state", + "action": "relocate " + "x" * 590, + "thought": "inspect " + "y" * 590, + "accessibility_tree": "tree " + "z" * 590, + "screenshot": "unused.png", + } + for index in range(10) + ] + + memory.insert(long_trajectory) + + assert len(runtime.memories) == 2 + l0_text, l1_text = (str(entry["text"]) for entry in runtime.memories) + assert len(l0_text.encode()) <= 1_024 + assert len(l1_text.encode()) == 8_192 + assert l1_text.startswith("LongMemEval-V2 deterministic L1 trajectory summary") + + +def test_query_returns_upstream_text_items_and_records_citations(tmp_path: Path) -> None: + runtime = FakeRuntime() + memory = adapter(tmp_path, runtime, search_mode="fts", search_limit=4) + memory.set_query_context(query_invocation_id="query-7") + + result = memory.query("Which assignment group should I use?", "question.png") + metadata = memory.post_query_hook( + query="Which assignment group should I use?", + query_image="question.png", + memory_context=result, + ) + + assert result == [{"type": "text", "value": "Use the Network assignment group."}] + assert runtime.searches == [ + { + "scope_id": "benchmark-run", + "query": "Which assignment group should I use?", + "limit": 4, + "mode": "fts", + } + ] + [event] = audit_events(tmp_path) + assert event["query_invocation_id"] == "query-7" + assert event["query_image_present"] is True + assert event["requested_mode"] == "fts" + assert event["actual_mode"] == "fts" + assert event["result_count"] == 1 + assert event["citations"][0]["entry_id"] == "entry-2" + assert "question.png" not in json.dumps(event) + assert set(event["timings_ms"]) == {"search", "format", "total"} + assert metadata == { + "query_invocation_id": "query-7", + "result_count": 1, + "retrieval_strategy": "memory-search", + "task_lens": None, + "requested_mode": "fts", + "actual_mode": "fts", + "citations": event["citations"], + "timings_ms": event["timings_ms"], + } + + +def test_failed_query_is_audited_without_question_or_image_content(tmp_path: Path) -> None: + class FailingRuntime(FakeRuntime): + def search_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + raise PowerContextMemoryAdapterError("unavailable") + + memory = adapter(tmp_path, FailingRuntime()) + + with pytest.raises(PowerContextMemoryAdapterError, match="unavailable"): + memory.query("secret-looking benchmark question", "private/image.png") + + [event] = audit_events(tmp_path) + assert event["status"] == "failed" + assert event["failure_type"] == "PowerContextMemoryAdapterError" + serialized = json.dumps(event) + assert "secret-looking" not in serialized + assert "private/image.png" not in serialized + + +def test_rejects_a_search_mode_outside_the_public_contract(tmp_path: Path) -> None: + with pytest.raises(PowerContextMemoryAdapterError, match="search_mode must be one of"): + adapter(tmp_path, FakeRuntime(), search_mode="semantic") + + with pytest.raises(PowerContextMemoryAdapterError, match="search_mode must be one of"): + adapter(tmp_path, FakeRuntime(), search_mode="hybrid-fts") + + +def test_hybrid_query_requests_and_records_the_executed_mode(tmp_path: Path) -> None: + runtime = FakeRuntime() + memory = adapter(tmp_path, runtime, search_mode="hybrid") + + result = memory.query("Which assignment group should I use?") + metadata = memory.post_query_hook( + query="Which assignment group should I use?", + query_image=None, + memory_context=result, + ) + + assert runtime.searches[0]["mode"] == "hybrid" + [event] = audit_events(tmp_path) + assert event["status"] == "succeeded" + assert event["requested_mode"] == "hybrid" + assert event["actual_mode"] == "hybrid" + assert metadata is not None + assert metadata["requested_mode"] == "hybrid" + assert metadata["actual_mode"] == "hybrid" + + +def test_query_time_compact_uses_public_prepared_context_and_records_strategy(tmp_path: Path) -> None: + runtime = FakeRuntime() + memory = adapter( + tmp_path, + runtime, + query_strategy="prepared-context", + prepared_context_max_bytes=4_096, + ) + + result = memory.query("Which assignment group should I use?") + metadata = memory.post_query_hook( + query="Which assignment group should I use?", + query_image=None, + memory_context=result, + ) + + assert result == [{"type": "text", "value": "Prepared compact context."}] + assert runtime.searches == [] + assert metadata is not None + assert metadata["retrieval_strategy"] == "prepared-context" + assert metadata["requested_mode"] is None + assert metadata["actual_mode"] is None + [event] = audit_events(tmp_path) + assert event["retrieval_strategy"] == "prepared-context" + + +def test_query_time_compact_rejects_a_response_over_its_byte_budget(tmp_path: Path) -> None: + class OversizedPreparedContextRuntime(FakeRuntime): + def prepare_context(self, payload: Mapping[str, object]) -> Mapping[str, object]: + return { + "schema": "powercontext.prepared-context.v1", + "status": "ready", + "content": "x" * 513, + "content_bytes": 513, + } + + memory = adapter( + tmp_path, + OversizedPreparedContextRuntime(), + query_strategy="prepared-context", + prepared_context_max_bytes=512, + ) + + with pytest.raises(PowerContextMemoryAdapterError, match="exceeded the requested byte budget"): + memory.query("Which assignment group should I use?") + + +def test_task_lens_projects_only_question_keywords_into_the_public_search(tmp_path: Path) -> None: + runtime = FakeRuntime() + memory = adapter(tmp_path, runtime, search_mode="fts", task_lens="question-keywords-v1") + + memory.query("Which assignment group should I use for the Network incident?") + + assert runtime.searches[0]["query"] == "assignment group use network incident" + [event] = audit_events(tmp_path) + assert event["task_lens"] == "question-keywords-v1" + assert event["query_sha256"] != event["retrieval_query_sha256"] + + +def test_fts_query_fails_closed_when_the_server_executed_a_different_mode(tmp_path: Path) -> None: + class MisreportingHybridRuntime(FakeRuntime): + def search_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + response = dict(super().search_memory(payload)) + response["mode"] = "hybrid" + return response + + memory = adapter(tmp_path, MisreportingHybridRuntime(), search_mode="fts") + + with pytest.raises(PowerContextMemoryModeError, match="reported mode 'hybrid'"): + memory.query("Which assignment group should I use?") + + [event] = audit_events(tmp_path) + assert event["status"] == "failed" + assert event["failure_type"] == "PowerContextMemoryModeError" + assert event["requested_mode"] == "fts" + assert event["actual_mode"] == "hybrid" + + +def test_hybrid_query_fails_closed_when_the_server_executed_a_different_mode(tmp_path: Path) -> None: + class SilentFtsFallbackRuntime(FakeRuntime): + def search_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + response = dict(super().search_memory(payload)) + response["mode"] = "fts" + return response + + memory = adapter(tmp_path, SilentFtsFallbackRuntime(), search_mode="hybrid") + + with pytest.raises(PowerContextMemoryModeError, match="reported mode 'fts'"): + memory.query("Which assignment group should I use?") + + [event] = audit_events(tmp_path) + assert event["status"] == "failed" + assert event["failure_type"] == "PowerContextMemoryModeError" + assert event["requested_mode"] == "hybrid" + assert event["actual_mode"] == "fts" + + +def test_an_explicit_mode_fails_closed_when_the_server_reports_no_mode(tmp_path: Path) -> None: + class UnreportedModeRuntime(FakeRuntime): + def search_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + response = dict(super().search_memory(payload)) + del response["mode"] + return response + + memory = adapter(tmp_path, UnreportedModeRuntime(), search_mode="hybrid") + + with pytest.raises(PowerContextMemoryModeError, match="reported mode None"): + memory.query("Which assignment group should I use?") + + [event] = audit_events(tmp_path) + assert event["status"] == "failed" + assert event["requested_mode"] == "hybrid" + assert event["actual_mode"] is None + + +def test_auto_query_accepts_the_server_chosen_mode(tmp_path: Path) -> None: + class VectorAutoRuntime(FakeRuntime): + def search_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + response = dict(super().search_memory(payload)) + response["mode"] = "vector" + return response + + memory = adapter(tmp_path, VectorAutoRuntime()) # the default mode is auto + + result = memory.query("Which assignment group should I use?") + + assert result == [{"type": "text", "value": "Use the Network assignment group."}] + [event] = audit_events(tmp_path) + assert event["status"] == "succeeded" + assert event["requested_mode"] == "auto" + assert event["actual_mode"] == "vector" + + +def test_duplicate_insert_fails_closed_and_records_the_attempt(tmp_path: Path) -> None: + runtime = FakeRuntime() + memory = adapter(tmp_path, runtime) + memory.insert(trajectory()) + + with pytest.raises(PowerContextMemoryAdapterError, match="duplicate trajectory"): + memory.insert(trajectory()) + + events = audit_events(tmp_path) + assert [event["status"] for event in events] == ["succeeded", "failed"] + assert len(runtime.captures) == 1 + + +def test_partial_ingest_failure_preserves_completed_chunk_citations(tmp_path: Path) -> None: + class PartiallyFailingRuntime(FakeRuntime): + def remember_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + raise PowerContextMemoryAdapterError("memory unavailable") + + runtime = PartiallyFailingRuntime() + memory = adapter(tmp_path, runtime, source_chunk_bytes=512) + + with pytest.raises(PowerContextMemoryAdapterError, match="memory unavailable"): + memory.insert(trajectory()) + + [event] = audit_events(tmp_path) + assert event["status"] == "failed" + assert len(event["source_refs"]) == len(runtime.captures) > 1 + assert event["memory_citations"] == [] + + +def test_http_runtime_rejects_credentials_and_plaintext_remote_hosts() -> None: + with pytest.raises(PowerContextMemoryAdapterError, match="credentials"): + PowerContextHTTPRuntime("https://token@example.test", token=None, timeout_seconds=1) + with pytest.raises(PowerContextMemoryAdapterError, match="unencrypted non-loopback"): + PowerContextHTTPRuntime("http://example.test", token=None, timeout_seconds=1) + with pytest.raises(PowerContextMemoryAdapterError, match="query or fragment"): + PowerContextHTTPRuntime("https://example.test/?token=secret", token=None, timeout_seconds=1) + + PowerContextHTTPRuntime("http://127.0.0.1:8765", token=None, timeout_seconds=1) diff --git a/evaluation/tests/unit/test_longmemeval_v2_arms.py b/evaluation/tests/unit/test_longmemeval_v2_arms.py new file mode 100644 index 0000000000..401451ee92 --- /dev/null +++ b/evaluation/tests/unit/test_longmemeval_v2_arms.py @@ -0,0 +1,242 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import dataclasses +from typing import cast + +import pytest + +from powercontext_eval.benchmarks.longmemeval_v2.arms import ( + CURRENT_MEMORY_FTS, + CURRENT_MEMORY_HYBRID, + DEFAULT_EXPERIMENT_ARM_ID, + EXPERIMENT_ARMS, + QUERY_TIME_COMPACT, + TASK_LENSED_SELECTION, + WRITE_TIME_L0_L1, + ExperimentArm, + ExperimentArmError, + arm_manifest_record, + ensure_comparable_experiment_runs, + get_experiment_arm, + resolve_experiment_arm, + supported_experiment_arm_ids, +) + + +def test_registry_contains_exactly_the_implemented_arms() -> None: + assert EXPERIMENT_ARMS == ( + CURRENT_MEMORY_FTS, + CURRENT_MEMORY_HYBRID, + QUERY_TIME_COMPACT, + WRITE_TIME_L0_L1, + TASK_LENSED_SELECTION, + ) + assert supported_experiment_arm_ids() == ( + "current-memory-fts-v1", + "current-memory-hybrid-v1", + "query-time-compact-v1", + "write-time-l0-l1-v1", + "task-lensed-selection-v1", + ) + assert DEFAULT_EXPERIMENT_ARM_ID == "current-memory-fts-v1" + + +def test_the_two_registered_arms_differ_only_in_identity_and_search_mode() -> None: + """A fair comparison requires every behavioural knob outside the arm to be equal.""" + + fts_fields = dataclasses.asdict(CURRENT_MEMORY_FTS) + hybrid_fields = dataclasses.asdict(CURRENT_MEMORY_HYBRID) + assert fts_fields.pop("arm_id") != hybrid_fields.pop("arm_id") + assert fts_fields.pop("search_mode") == "fts" + assert hybrid_fields.pop("search_mode") == "hybrid" + assert fts_fields == hybrid_fields + + +def test_get_experiment_arm_returns_the_registered_arm() -> None: + assert get_experiment_arm("current-memory-fts-v1") is CURRENT_MEMORY_FTS + assert get_experiment_arm("current-memory-hybrid-v1") is CURRENT_MEMORY_HYBRID + assert get_experiment_arm("query-time-compact-v1") is QUERY_TIME_COMPACT + assert get_experiment_arm("write-time-l0-l1-v1") is WRITE_TIME_L0_L1 + assert get_experiment_arm("task-lensed-selection-v1") is TASK_LENSED_SELECTION + + +@pytest.mark.parametrize("arm_id", ["", " ", "l0-persistent-v1", "temporal-recency-v1", "task-lens-v1"]) +def test_get_experiment_arm_rejects_unknown_ids_with_the_supported_list(arm_id: str) -> None: + with pytest.raises(ExperimentArmError, match="supported arms"): + get_experiment_arm(arm_id) + + +def test_resolve_experiment_arm_defaults_to_the_fts_arm() -> None: + assert resolve_experiment_arm(None) is CURRENT_MEMORY_FTS + assert resolve_experiment_arm("current-memory-hybrid-v1") is CURRENT_MEMORY_HYBRID + assert resolve_experiment_arm(CURRENT_MEMORY_HYBRID) is CURRENT_MEMORY_HYBRID + + +def test_arm_manifest_record_matches_the_documented_shape() -> None: + assert arm_manifest_record(CURRENT_MEMORY_HYBRID) == { + "id": "current-memory-hybrid-v1", + "retrieval_strategy": "memory-search", + "search_mode": "hybrid", + "memory_projection": "deterministic-compact-v1", + "query_projection": "question-text-v1", + "prepared_context_max_bytes": None, + "temporal_filter": None, + "task_lens": None, + } + + +def test_query_time_compact_arm_uses_prepared_context_without_changing_ingestion() -> None: + assert arm_manifest_record(QUERY_TIME_COMPACT) == { + "id": "query-time-compact-v1", + "retrieval_strategy": "prepared-context", + "search_mode": None, + "memory_projection": "deterministic-compact-v1", + "query_projection": "powercontext-prepared-context-v1", + "prepared_context_max_bytes": 8_000, + "temporal_filter": None, + "task_lens": None, + } + + +def test_write_time_l0_l1_arm_changes_only_the_declared_memory_projection() -> None: + record = arm_manifest_record(WRITE_TIME_L0_L1) + assert record["id"] == "write-time-l0-l1-v1" + assert record["retrieval_strategy"] == "memory-search" + assert record["search_mode"] == "fts" + assert record["memory_projection"] == "deterministic-l0-l1-v1" + assert record["query_projection"] == "question-text-v1" + + +def test_task_lensed_arm_declares_its_deterministic_query_projection() -> None: + record = arm_manifest_record(TASK_LENSED_SELECTION) + assert record["id"] == "task-lensed-selection-v1" + assert record["search_mode"] == "fts" + assert record["memory_projection"] == "deterministic-compact-v1" + assert record["query_projection"] == "deterministic-keyword-lens-v1" + assert record["task_lens"] == "question-keywords-v1" + + +def comparable_manifest(arm: ExperimentArm, **overrides: object) -> dict[str, object]: + manifest: dict[str, object] = { + "schema": "powercontext.longmemeval-v2-smoke-run.v1", + "classification": "smoke-subset", + "run_id": "run-a", + "started_at": "2026-09-20T00:00:00+00:00", + "experiment_arm": arm_manifest_record(arm), + "inputs": { + "data_root": "D:/data", + "dataset_lock": {"path": "D:/locks/a.json", "content_sha256": "lock-digest"}, + "smoke_manifest": {"path": "D:/locks/b.json", "content_sha256": "smoke-digest"}, + "harness": {"root": "D:/harness", "commit": "2cc8c540", "python": "D:/python"}, + }, + "processor": {"model": "Qwen/Qwen3.5-9B", "revision": "processor-sha"}, + "memory_context_max_tokens": 200_000, + "powercontext": { + "base_url": "http://127.0.0.1:18765", + "token_env": "POWERCONTEXT_TOKEN", + "search_mode": arm.search_mode, + "search_limit": 10, + "timeout_seconds": 30.0, + }, + "reader": None, + "judge": None, + "revisions": {"powercontext": "pc-sha", "integration": "integration-sha"}, + } + manifest.update(overrides) + return manifest + + +def test_runs_that_differ_only_by_arm_are_comparable() -> None: + ensure_comparable_experiment_runs( + comparable_manifest(CURRENT_MEMORY_FTS), + comparable_manifest(CURRENT_MEMORY_HYBRID), + ) + + +def test_runs_with_missing_comparison_fields_are_refused() -> None: + manifest_a = comparable_manifest(CURRENT_MEMORY_FTS) + base = comparable_manifest(CURRENT_MEMORY_HYBRID) + inputs = dict(cast("dict[str, object]", base["inputs"])) + del inputs["dataset_lock"] + manifest_b = {**base, "inputs": inputs} + + with pytest.raises(ExperimentArmError, match=r"inputs\.dataset_lock\.content_sha256 is missing"): + ensure_comparable_experiment_runs(manifest_a, manifest_b) + + revisions = dict(cast("dict[str, object]", base["revisions"])) + del revisions["integration"] + manifest_c = {**base, "revisions": revisions} + with pytest.raises(ExperimentArmError, match=r"revisions\.integration is missing"): + ensure_comparable_experiment_runs(manifest_a, manifest_c) + + +def test_runs_without_a_registered_arm_record_are_refused() -> None: + manifest_a = comparable_manifest(CURRENT_MEMORY_FTS) + + without_arm = { + key: value for key, value in comparable_manifest(CURRENT_MEMORY_HYBRID).items() if key != "experiment_arm" + } + with pytest.raises(ExperimentArmError, match="has no experiment_arm record"): + ensure_comparable_experiment_runs(manifest_a, without_arm) + + tampered_arm = dict(cast("dict[str, object]", comparable_manifest(CURRENT_MEMORY_HYBRID)["experiment_arm"])) + tampered_arm["search_mode"] = "vector" + manifest_b = {**comparable_manifest(CURRENT_MEMORY_HYBRID), "experiment_arm": tampered_arm} + with pytest.raises(ExperimentArmError, match="unregistered experiment_arm record"): + ensure_comparable_experiment_runs(manifest_a, manifest_b) + + +def test_runs_that_differ_beyond_the_arm_are_refused() -> None: + manifest_a = comparable_manifest(CURRENT_MEMORY_FTS) + manifest_b = comparable_manifest(CURRENT_MEMORY_HYBRID) + manifest_b["inputs"] = { + "data_root": "D:/data", + "dataset_lock": {"path": "D:/locks/other.json", "content_sha256": "other-lock-digest"}, + "smoke_manifest": {"path": "D:/locks/b.json", "content_sha256": "smoke-digest"}, + "harness": {"root": "D:/harness", "commit": "2cc8c540", "python": "D:/python"}, + } + with pytest.raises(ExperimentArmError, match=r"inputs\.dataset_lock\.content_sha256"): + ensure_comparable_experiment_runs(manifest_a, manifest_b) + + +def test_a_different_search_limit_or_processor_refuses_the_comparison() -> None: + manifest_a = comparable_manifest(CURRENT_MEMORY_FTS) + + smaller_budget = comparable_manifest(CURRENT_MEMORY_HYBRID, memory_context_max_tokens=100_000) + with pytest.raises(ExperimentArmError, match="memory_context_max_tokens"): + ensure_comparable_experiment_runs(manifest_a, smaller_budget) + + other_limit = comparable_manifest(CURRENT_MEMORY_HYBRID) + other_limit["powercontext"] = { + "base_url": "http://127.0.0.1:18765", + "token_env": "POWERCONTEXT_TOKEN", + "search_mode": "hybrid", + "search_limit": 5, + "timeout_seconds": 30.0, + } + with pytest.raises(ExperimentArmError, match=r"powercontext\.search_limit"): + ensure_comparable_experiment_runs(manifest_a, other_limit) + + other_reader = comparable_manifest( + CURRENT_MEMORY_HYBRID, reader={"provider": "deepseek-openai", "model": "deepseek-chat"} + ) + with pytest.raises(ExperimentArmError, match="reader"): + ensure_comparable_experiment_runs(manifest_a, other_reader) + + other_powercontext_revision = comparable_manifest( + CURRENT_MEMORY_HYBRID, revisions={"powercontext": "other-sha", "integration": "integration-sha"} + ) + with pytest.raises(ExperimentArmError, match=r"revisions\.powercontext"): + ensure_comparable_experiment_runs(manifest_a, other_powercontext_revision) diff --git a/evaluation/tests/unit/test_longmemeval_v2_catalog.py b/evaluation/tests/unit/test_longmemeval_v2_catalog.py index 1b54f4e2c6..a8d6147ea1 100644 --- a/evaluation/tests/unit/test_longmemeval_v2_catalog.py +++ b/evaluation/tests/unit/test_longmemeval_v2_catalog.py @@ -25,11 +25,15 @@ import powercontext_eval.benchmarks.longmemeval_v2.smoke as longmemeval_smoke from powercontext_eval.benchmarks.longmemeval_v2.catalog import ( RUN_INPUT_MANIFEST_SCHEMA, + RUN_SUBSET_SCHEMA, SMOKE_MANIFEST_SCHEMA, UPSTREAM_HARNESS_COMMIT, LongMemEvalV2Catalog, LongMemEvalV2CatalogError, + LongMemEvalV2EnvironmentError, + LongMemEvalV2InputError, SmokeCase, + SmokeSelection, load_dataset_lock, load_smoke_manifest, validate_harness_checkout, @@ -171,7 +175,7 @@ def test_catalog_rejects_a_cross_domain_haystack(tmp_path: Path) -> None: def test_smoke_selection_requires_all_published_abilities(tmp_path: Path) -> None: catalog = LongMemEvalV2Catalog.load(data_root(tmp_path), tier="small") - with pytest.raises(LongMemEvalV2CatalogError, match="missing published abilities"): + with pytest.raises(LongMemEvalV2InputError, match="missing published abilities"): catalog.select_smoke((SmokeCase(question_id="q-static", ability="static_state"),)) @@ -222,7 +226,7 @@ def test_dataset_lock_pins_exact_file_digests(tmp_path: Path) -> None: LongMemEvalV2Catalog.load(root, tier=loaded.tier, expected_digests=loaded.file_digests) (root / "questions.jsonl").write_text("{}\n", encoding="utf-8") - with pytest.raises(LongMemEvalV2CatalogError, match="SHA-256 mismatch"): + with pytest.raises(LongMemEvalV2EnvironmentError, match="SHA-256 mismatch"): LongMemEvalV2Catalog.load(root, tier=loaded.tier, expected_digests=loaded.file_digests) @@ -253,7 +257,7 @@ def test_harness_checkout_rejects_a_different_revision(monkeypatch, tmp_path: Pa harness, _revision = harness_checkout(tmp_path) monkeypatch.setattr(longmemeval_catalog, "UPSTREAM_HARNESS_COMMIT", "0" * 40) - with pytest.raises(LongMemEvalV2CatalogError, match="harness checkout must be"): + with pytest.raises(LongMemEvalV2EnvironmentError, match="harness checkout must be"): validate_harness_checkout(harness) @@ -266,7 +270,7 @@ def test_harness_checkout_rejects_tracked_edits(monkeypatch, tmp_path: Path, sta if staged: subprocess.run(("git", "-C", str(harness), "add", "evaluation/harness.py"), check=True) - with pytest.raises(LongMemEvalV2CatalogError, match="tracked changes"): + with pytest.raises(LongMemEvalV2EnvironmentError, match="tracked changes"): validate_harness_checkout(harness) @@ -290,14 +294,17 @@ def test_prepare_smoke_run_writes_non_overwritable_provenance(tmp_path: Path) -> output_dir=output, ) manifest = json.loads((output / "manifest.json").read_text(encoding="utf-8")) + subset = json.loads((output / "subset.json").read_text(encoding="utf-8")) assert manifest["schema"] == RUN_INPUT_MANIFEST_SCHEMA assert manifest["classification"] == "smoke-subset" + assert subset["schema"] == RUN_SUBSET_SCHEMA + assert subset["classification"] == "smoke-subset" assert manifest["upstream"]["harness_commit"] == revision assert manifest["dataset"]["revision"] == "fixture-data-revision" assert manifest["dataset_lock"]["content_sha256"] == hashlib.sha256(lock.read_bytes()).hexdigest() assert manifest["smoke_manifest"]["content_sha256"] == hashlib.sha256(smoke.read_bytes()).hexdigest() assert "path_sha256" not in manifest["smoke_manifest"] - with pytest.raises(LongMemEvalV2CatalogError, match="Refusing to overwrite"): + with pytest.raises(LongMemEvalV2EnvironmentError, match="Refusing to overwrite"): prepare_smoke_run( data_root=root, dataset_lock=lock, @@ -307,3 +314,26 @@ def test_prepare_smoke_run_writes_non_overwritable_provenance(tmp_path: Path) -> ) finally: monkeypatch.undo() + + +def test_prepare_smoke_run_serializes_the_validated_selection(monkeypatch, tmp_path: Path) -> None: + root = data_root(tmp_path) + smoke = smoke_manifest(tmp_path / "smoke.json") + harness, revision = harness_checkout(tmp_path) + lock = dataset_lock(tmp_path / "dataset-lock.json", root, harness_commit=revision) + output = tmp_path / "run" + validated = SmokeSelection(tier="small", cases=(SmokeCase(question_id="q-static", ability="static_state"),)) + monkeypatch.setattr(longmemeval_catalog, "UPSTREAM_HARNESS_COMMIT", revision) + monkeypatch.setattr(longmemeval_smoke, "UPSTREAM_HARNESS_COMMIT", revision) + monkeypatch.setattr(LongMemEvalV2Catalog, "select_smoke", lambda self, cases: validated) + + prepare_smoke_run( + data_root=root, + dataset_lock=lock, + harness_root=harness, + smoke_manifest=smoke, + output_dir=output, + ) + + subset = json.loads((output / "subset.json").read_text(encoding="utf-8")) + assert subset["cases"] == [{"question_id": "q-static", "ability": "static_state"}] diff --git a/evaluation/tests/unit/test_longmemeval_v2_cli.py b/evaluation/tests/unit/test_longmemeval_v2_cli.py new file mode 100644 index 0000000000..fd64e232ed --- /dev/null +++ b/evaluation/tests/unit/test_longmemeval_v2_cli.py @@ -0,0 +1,921 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import json +from pathlib import Path + +import pytest +from typer.testing import CliRunner + +from powercontext_eval.benchmarks.longmemeval_v2.costs import ModelPricePolicy +from powercontext_eval.benchmarks.longmemeval_v2.prepare_smoke import PreparedPromptRun +from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import ReaderSmokeRun +from powercontext_eval.benchmarks.longmemeval_v2.replay_score import ReplayScoreRun +from powercontext_eval.benchmarks.longmemeval_v2.report import ReportError, ReportRun +from powercontext_eval.benchmarks.longmemeval_v2.retrieval_smoke import ( + RetrievalCapabilityError, + RetrievalSmokeRun, +) +from powercontext_eval.benchmarks.longmemeval_v2.run_smoke import RunSmokeError, SmokeRunResult +from powercontext_eval.benchmarks.longmemeval_v2.score_smoke import ScoreSmokeRun +from powercontext_eval.cli import app + + +def test_longmemeval_v2_retrieval_smoke_runs_without_reader_or_judge( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + calls: list[dict[str, object]] = [] + + def run(**kwargs: object) -> RetrievalSmokeRun: + calls.append(kwargs) + output = tmp_path / "output" + return RetrievalSmokeRun( + output_dir=output, + manifest_path=output / "retrieval-manifest.json", + results_path=output / "retrieval-results.jsonl", + failures_path=output / "failures.jsonl", + summary_path=output / "summary.json", + audit_path=output / "adapter-audit.jsonl", + ) + + monkeypatch.setattr("powercontext_eval.cli.run_retrieval_smoke", run) + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "retrieval-smoke", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--harness-root", + "/harness", + "--smoke-manifest", + "/smoke.json", + "--output-dir", + "/output", + "--run-id", + "retrieval-1", + "--powercontext-revision", + "pc-sha", + "--integration-revision", + "adapter-sha", + ], + ) + + assert result.exit_code == 0, result.output + assert '"classification": "smoke-subset-retrieval-only"' in result.output + assert calls == [ + { + "data_root": Path("/data"), + "dataset_lock": Path("/dataset-lock.json"), + "harness_root": Path("/harness"), + "smoke_manifest": Path("/smoke.json"), + "output_dir": Path("/output"), + "run_id": "retrieval-1", + "powercontext_revision": "pc-sha", + "integration_revision": "adapter-sha", + "base_url": "http://127.0.0.1:8000", + "token_env": "POWERCONTEXT_TOKEN", + "experiment_arm": "current-memory-fts-v1", + "search_limit": 10, + "timeout_seconds": 30.0, + } + ] + + +def test_longmemeval_v2_prepare_smoke_runs_without_reader_or_judge( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + calls: list[dict[str, object]] = [] + + def prepare(**kwargs: object) -> PreparedPromptRun: + calls.append(kwargs) + output = tmp_path / "output" + return PreparedPromptRun( + output_dir=output, + manifest_path=output / "prepare-manifest.json", + prompts_path=output / "prepared-prompts.jsonl", + failures_path=output / "prepare-failures.jsonl", + summary_path=output / "prepare-summary.json", + ) + + monkeypatch.setattr("powercontext_eval.cli.prepare_reader_inputs_smoke", prepare) + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "prepare-smoke", + "--retrieval-dir", + "/retrieval", + "--harness-root", + "/harness", + "--harness-python", + "/harness/python", + "--output-dir", + "/output", + "--processor-revision", + "processor-sha", + ], + ) + + assert result.exit_code == 0, result.output + assert '"classification": "smoke-subset-prepare-only"' in result.output + assert calls == [ + { + "retrieval_dir": Path("/retrieval"), + "harness_root": Path("/harness"), + "harness_python": Path("/harness/python"), + "output_dir": Path("/output"), + "processor_model": "Qwen/Qwen3.5-9B", + "processor_revision": "processor-sha", + "memory_context_max_tokens": 200_000, + } + ] + + +def test_longmemeval_v2_reader_smoke_uses_environment_reference_without_secret( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + calls: list[dict[str, object]] = [] + + def run(**kwargs: object) -> ReaderSmokeRun: + calls.append(kwargs) + output = tmp_path / "output" + return ReaderSmokeRun( + output_dir=output, + manifest_path=output / "reader-manifest.json", + outputs_path=output / "reader-outputs.jsonl", + failures_path=output / "reader-failures.jsonl", + summary_path=output / "reader-summary.json", + ) + + monkeypatch.setattr("powercontext_eval.cli.run_reader_smoke", run) + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "reader-smoke", + "--prepared-dir", + "/prepared", + "--output-dir", + "/output", + ], + ) + + assert result.exit_code == 0, result.output + assert '"classification": "smoke-subset-reader-only"' in result.output + assert calls == [ + { + "prepared_dir": Path("/prepared"), + "output_dir": Path("/output"), + "provider": "anthropic-compatible", + "model": None, + "base_url": None, + "base_url_env": "ANTHROPIC_BASE_URL", + "token_env": None, + "max_tokens": 512, + "temperature": 0.0, + "timeout_seconds": 120.0, + "max_questions": None, + "price_policy": None, + } + ] + + +def test_longmemeval_v2_score_smoke_uses_local_reader_artifacts_and_judge_environment( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + calls: list[dict[str, object]] = [] + + def run(**kwargs: object) -> ScoreSmokeRun: + calls.append(kwargs) + output = tmp_path / "output" + return ScoreSmokeRun( + output_dir=output, + manifest_path=output / "score-manifest.json", + inputs_path=output / "scoring-inputs.local.jsonl", + results_path=output / "per-question.jsonl", + judge_outputs_path=output / "judge-outputs.jsonl", + failures_path=output / "score-failures.jsonl", + summary_path=output / "score-summary.json", + ) + + monkeypatch.setattr("powercontext_eval.cli.run_score_smoke", run) + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "score-smoke", + "--reader-dir", + "/reader", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--smoke-manifest", + "/smoke.json", + "--harness-root", + "/harness", + "--output-dir", + "/output", + ], + ) + + assert result.exit_code == 0, result.output + assert '"classification": "smoke-subset-score-only"' in result.output + assert calls == [ + { + "reader_dir": Path("/reader"), + "data_root": Path("/data"), + "dataset_lock": Path("/dataset-lock.json"), + "smoke_manifest": Path("/smoke.json"), + "harness_root": Path("/harness"), + "output_dir": Path("/output"), + "judge_model": "deepseek-flash", + "judge_token_env": "DEEPSEEK_API_KEY", + "judge_base_url": "https://api.deepseek.com", + "judge_max_tokens": 256, + "judge_temperature": 0.0, + "judge_timeout_seconds": 120.0, + "judge_price_policy": None, + } + ] + + +def test_longmemeval_v2_replay_score_uses_only_local_score_artifacts( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + calls: list[dict[str, object]] = [] + + def replay(**kwargs: object) -> ReplayScoreRun: + calls.append(kwargs) + output = tmp_path / "output" + return ReplayScoreRun( + output_dir=output, + manifest_path=output / "replay-manifest.json", + results_path=output / "replay-per-question.jsonl", + failures_path=output / "replay-failures.jsonl", + summary_path=output / "replay-summary.json", + ) + + monkeypatch.setattr("powercontext_eval.cli.replay_score_smoke", replay) + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "replay-score", + "--score-dir", + "/score", + "--harness-root", + "/harness", + "--output-dir", + "/output", + ], + ) + + assert result.exit_code == 0, result.output + assert '"classification": "smoke-subset-score-replay"' in result.output + assert calls == [{"score_dir": Path("/score"), "harness_root": Path("/harness"), "output_dir": Path("/output")}] + + +def test_longmemeval_v2_run_smoke_forwards_every_stage_option(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None: + calls: list[dict[str, object]] = [] + + def run(**kwargs: object) -> SmokeRunResult: + calls.append(kwargs) + output = tmp_path / "output" + return SmokeRunResult( + output_dir=output, + manifest_path=output / "run-manifest.json", + summary_path=output / "run-summary.json", + failures_path=output / "failures.jsonl", + status="completed", + completed_phases=("preflight", "retrieval", "prepare", "reader", "score", "replay"), + skipped_phases=(), + failed_phase=None, + ) + + monkeypatch.setattr("powercontext_eval.cli.run_smoke", run) + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "run-smoke", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--smoke-manifest", + "/smoke.json", + "--harness-root", + "/harness", + "--harness-python", + "/harness/python", + "--processor-revision", + "processor-sha", + "--output-dir", + "/output", + "--powercontext-revision", + "pc-sha", + "--integration-revision", + "integration-sha", + ], + ) + + assert result.exit_code == 0, result.output + assert '"status": "completed"' in result.output + assert calls == [ + { + "data_root": Path("/data"), + "dataset_lock": Path("/dataset-lock.json"), + "smoke_manifest": Path("/smoke.json"), + "harness_root": Path("/harness"), + "harness_python": Path("/harness/python"), + "processor_revision": "processor-sha", + "output_dir": Path("/output"), + "powercontext_revision": "pc-sha", + "integration_revision": "integration-sha", + "run_id": None, + "processor_model": "Qwen/Qwen3.5-9B", + "memory_context_max_tokens": 200_000, + "powercontext_base_url": "http://127.0.0.1:8000", + "powercontext_token_env": "POWERCONTEXT_TOKEN", + "experiment_arm": "current-memory-fts-v1", + "search_limit": 10, + "timeout_seconds": 30.0, + "reader_provider": "deepseek-openai", + "reader_model": None, + "reader_base_url": None, + "reader_base_url_env": "ANTHROPIC_BASE_URL", + "reader_token_env": None, + "reader_max_tokens": 512, + "reader_temperature": 0.0, + "reader_timeout_seconds": 120.0, + "judge_provider": "deepseek-openai", + "judge_model": "deepseek-flash", + "judge_token_env": "DEEPSEEK_API_KEY", + "judge_base_url": "https://api.deepseek.com", + "judge_max_tokens": 256, + "judge_temperature": 0.0, + "judge_timeout_seconds": 120.0, + "reader_price_policy": None, + "judge_price_policy": None, + "skip_reader": False, + "skip_score": False, + } + ] + + +def test_longmemeval_v2_run_smoke_exits_nonzero_without_a_complete_run( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + def run(**kwargs: object) -> SmokeRunResult: + output = tmp_path / "output" + return SmokeRunResult( + output_dir=output, + manifest_path=output / "run-manifest.json", + summary_path=output / "run-summary.json", + failures_path=output / "failures.jsonl", + status="failed", + completed_phases=("preflight", "retrieval"), + skipped_phases=("prepare", "reader", "score", "replay"), + failed_phase="prepare", + ) + + monkeypatch.setattr("powercontext_eval.cli.run_smoke", run) + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "run-smoke", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--smoke-manifest", + "/smoke.json", + "--harness-root", + "/harness", + "--harness-python", + "/harness/python", + "--processor-revision", + "processor-sha", + "--output-dir", + "/output", + "--powercontext-revision", + "pc-sha", + "--integration-revision", + "integration-sha", + "--skip-reader", + ], + ) + + assert result.exit_code == 1 + assert '"failed_phase": "prepare"' in result.output + assert '"status": "failed"' in result.output + + +def test_longmemeval_v2_run_smoke_reports_a_refused_output_directory( + monkeypatch: pytest.MonkeyPatch, +) -> None: + def run(**kwargs: object) -> SmokeRunResult: + raise RunSmokeError("Refusing to overwrite smoke run artifacts: /output") + + monkeypatch.setattr("powercontext_eval.cli.run_smoke", run) + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "run-smoke", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--smoke-manifest", + "/smoke.json", + "--harness-root", + "/harness", + "--harness-python", + "/harness/python", + "--processor-revision", + "processor-sha", + "--output-dir", + "/output", + "--powercontext-revision", + "pc-sha", + "--integration-revision", + "integration-sha", + ], + ) + + assert result.exit_code == 1 + assert "Refusing to overwrite smoke run artifacts" in result.output + + +def test_longmemeval_v2_run_smoke_forwards_a_selected_experiment_arm( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + calls: list[dict[str, object]] = [] + + def run(**kwargs: object) -> SmokeRunResult: + calls.append(kwargs) + output = tmp_path / "output" + return SmokeRunResult( + output_dir=output, + manifest_path=output / "run-manifest.json", + summary_path=output / "run-summary.json", + failures_path=output / "failures.jsonl", + status="partial", + completed_phases=("preflight", "retrieval", "prepare"), + skipped_phases=("reader", "score", "replay"), + failed_phase=None, + ) + + monkeypatch.setattr("powercontext_eval.cli.run_smoke", run) + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "run-smoke", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--smoke-manifest", + "/smoke.json", + "--harness-root", + "/harness", + "--harness-python", + "/harness/python", + "--processor-revision", + "processor-sha", + "--output-dir", + "/output", + "--powercontext-revision", + "pc-sha", + "--integration-revision", + "integration-sha", + "--skip-reader", + "--experiment-arm", + "current-memory-hybrid-v1", + ], + ) + + assert result.exit_code == 1, result.output # a partial model-free run exits non-zero by design + assert calls[0]["experiment_arm"] == "current-memory-hybrid-v1" + + +def test_longmemeval_v2_run_smoke_rejects_an_unknown_experiment_arm() -> None: + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "run-smoke", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--smoke-manifest", + "/smoke.json", + "--harness-root", + "/harness", + "--harness-python", + "/harness/python", + "--processor-revision", + "processor-sha", + "--output-dir", + "/output", + "--powercontext-revision", + "pc-sha", + "--integration-revision", + "integration-sha", + "--experiment-arm", + "l0-persistent-v1", + ], + ) + + assert result.exit_code == 1, result.output + assert "unknown experiment arm" in result.output + assert "current-memory-fts-v1" in result.output + assert "current-memory-hybrid-v1" in result.output + + +def test_longmemeval_v2_retrieval_smoke_rejects_an_unknown_experiment_arm() -> None: + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "retrieval-smoke", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--harness-root", + "/harness", + "--smoke-manifest", + "/smoke.json", + "--output-dir", + "/output", + "--run-id", + "retrieval-1", + "--powercontext-revision", + "pc-sha", + "--integration-revision", + "adapter-sha", + "--experiment-arm", + "temporal-recency-v1", + ], + ) + + assert result.exit_code == 2, result.output # an unknown arm is a parameter error + assert "unknown experiment arm" in result.output + + +def test_longmemeval_v2_retrieval_smoke_reports_a_capability_failure_as_a_runtime_error( + monkeypatch: pytest.MonkeyPatch, +) -> None: + def run(**kwargs: object) -> RetrievalSmokeRun: + raise RetrievalCapabilityError( + "PowerContext Server does not support hybrid Memory search " + "required by experiment arm current-memory-hybrid-v1" + ) + + monkeypatch.setattr("powercontext_eval.cli.run_retrieval_smoke", run) + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "retrieval-smoke", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--harness-root", + "/harness", + "--smoke-manifest", + "/smoke.json", + "--output-dir", + "/output", + "--run-id", + "retrieval-1", + "--powercontext-revision", + "pc-sha", + "--integration-revision", + "adapter-sha", + "--experiment-arm", + "current-memory-hybrid-v1", + ], + ) + + assert result.exit_code == 1, result.output # the environment, not the argument, is wrong + assert "does not support hybrid Memory search" in result.output + assert "Invalid value" not in result.output + + +def test_longmemeval_v2_report_reads_only_saved_run_artifacts(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None: + calls: list[dict[str, object]] = [] + + def report(**kwargs: object) -> ReportRun: + calls.append(kwargs) + return ReportRun(report_path=tmp_path / "report.json", markdown_path=tmp_path / "report.md") + + monkeypatch.setattr("powercontext_eval.cli.build_report", report) + result = CliRunner().invoke(app, ["longmemeval-v2", "report", "--run-dir", "/run"]) + + assert result.exit_code == 0, result.output + assert '"classification": "smoke-subset"' in result.output + assert calls == [{"run_dir": Path("/run"), "output_dir": None}] + + +def test_longmemeval_v2_report_exits_nonzero_for_a_refused_target(monkeypatch: pytest.MonkeyPatch) -> None: + def report(**kwargs: object) -> ReportRun: + raise ReportError("Refusing to overwrite report artifacts: /report") + + monkeypatch.setattr("powercontext_eval.cli.build_report", report) + result = CliRunner().invoke(app, ["longmemeval-v2", "report", "--run-dir", "/run", "--output-dir", "/report"]) + + assert result.exit_code == 1 + assert "Refusing to overwrite report artifacts" in result.output + + +PRICE_POLICY_JSON = json.dumps( + { + "provider": "deepseek-openai", + "model": "deepseek-flash", + "currency": "USD", + "input_cache_hit_price_per_million": 0.006, + "input_cache_miss_price_per_million": 0.3, + "output_price_per_million": 1.2, + "price_policy_revision": "deepseek-public-list-2026-09", + } +) + + +def test_longmemeval_v2_reader_smoke_forwards_an_explicit_price_policy( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + calls: list[dict[str, object]] = [] + + def run(**kwargs: object) -> ReaderSmokeRun: + calls.append(kwargs) + output = tmp_path / "output" + return ReaderSmokeRun( + output_dir=output, + manifest_path=output / "reader-manifest.json", + outputs_path=output / "reader-outputs.jsonl", + failures_path=output / "reader-failures.jsonl", + summary_path=output / "reader-summary.json", + ) + + monkeypatch.setattr("powercontext_eval.cli.run_reader_smoke", run) + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "reader-smoke", + "--prepared-dir", + "/prepared", + "--output-dir", + "/output", + "--price-policy", + PRICE_POLICY_JSON, + ], + ) + + assert result.exit_code == 0, result.output + [call] = calls + assert call["price_policy"] == ModelPricePolicy( + provider="deepseek-openai", + model="deepseek-flash", + currency="USD", + input_cache_hit_price_per_million=0.006, + input_cache_miss_price_per_million=0.3, + output_price_per_million=1.2, + price_policy_revision="deepseek-public-list-2026-09", + ) + + +def test_longmemeval_v2_score_smoke_forwards_an_explicit_judge_price_policy( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + calls: list[dict[str, object]] = [] + + def run(**kwargs: object) -> ScoreSmokeRun: + calls.append(kwargs) + output = tmp_path / "output" + return ScoreSmokeRun( + output_dir=output, + manifest_path=output / "score-manifest.json", + inputs_path=output / "scoring-inputs.local.jsonl", + results_path=output / "per-question.jsonl", + judge_outputs_path=output / "judge-outputs.jsonl", + failures_path=output / "score-failures.jsonl", + summary_path=output / "score-summary.json", + ) + + monkeypatch.setattr("powercontext_eval.cli.run_score_smoke", run) + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "score-smoke", + "--reader-dir", + "/reader", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--smoke-manifest", + "/smoke.json", + "--harness-root", + "/harness", + "--output-dir", + "/output", + "--judge-price-policy", + PRICE_POLICY_JSON, + ], + ) + + assert result.exit_code == 0, result.output + [call] = calls + assert call["judge_price_policy"] == ModelPricePolicy( + provider="deepseek-openai", + model="deepseek-flash", + currency="USD", + input_cache_hit_price_per_million=0.006, + input_cache_miss_price_per_million=0.3, + output_price_per_million=1.2, + price_policy_revision="deepseek-public-list-2026-09", + ) + + +def test_longmemeval_v2_run_smoke_forwards_separate_reader_and_judge_price_policies( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + calls: list[dict[str, object]] = [] + + def run(**kwargs: object) -> SmokeRunResult: + calls.append(kwargs) + output = tmp_path / "output" + return SmokeRunResult( + output_dir=output, + manifest_path=output / "run-manifest.json", + summary_path=output / "run-summary.json", + failures_path=output / "failures.jsonl", + status="completed", + completed_phases=("preflight", "retrieval", "prepare", "reader", "score", "replay"), + skipped_phases=(), + failed_phase=None, + ) + + monkeypatch.setattr("powercontext_eval.cli.run_smoke", run) + judge_policy_json = json.dumps({**json.loads(PRICE_POLICY_JSON), "model": "deepseek-v4-pro"}) + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "run-smoke", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--smoke-manifest", + "/smoke.json", + "--harness-root", + "/harness", + "--harness-python", + "/harness/python", + "--processor-revision", + "processor-sha", + "--output-dir", + "/output", + "--powercontext-revision", + "pc-sha", + "--integration-revision", + "integration-sha", + "--price-policy", + PRICE_POLICY_JSON, + "--judge-price-policy", + judge_policy_json, + ], + ) + + assert result.exit_code == 0, result.output + [call] = calls + reader_policy = call["reader_price_policy"] + judge_policy = call["judge_price_policy"] + assert isinstance(reader_policy, ModelPricePolicy) + assert isinstance(judge_policy, ModelPricePolicy) + assert reader_policy.model == "deepseek-flash" + assert judge_policy.model == "deepseek-v4-pro" + + +def test_longmemeval_v2_run_smoke_rejects_a_malformed_price_policy() -> None: + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "run-smoke", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--smoke-manifest", + "/smoke.json", + "--harness-root", + "/harness", + "--harness-python", + "/harness/python", + "--processor-revision", + "processor-sha", + "--output-dir", + "/output", + "--powercontext-revision", + "pc-sha", + "--integration-revision", + "integration-sha", + "--price-policy", + '{"provider": "deepseek-openai"}', + ], + ) + + assert result.exit_code != 0 + assert "must contain exactly" in result.output + + +def test_longmemeval_v2_run_smoke_rejects_a_non_usd_price_policy() -> None: + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "run-smoke", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--smoke-manifest", + "/smoke.json", + "--harness-root", + "/harness", + "--harness-python", + "/harness/python", + "--processor-revision", + "processor-sha", + "--output-dir", + "/output", + "--powercontext-revision", + "pc-sha", + "--integration-revision", + "integration-sha", + "--price-policy", + json.dumps({**json.loads(PRICE_POLICY_JSON), "currency": "CNY"}), + ], + ) + + assert result.exit_code != 0 + assert "currency must be USD" in result.output + + +def test_longmemeval_v2_run_smoke_rejects_a_non_json_price_policy() -> None: + result = CliRunner().invoke( + app, + [ + "longmemeval-v2", + "run-smoke", + "--data-root", + "/data", + "--dataset-lock", + "/dataset-lock.json", + "--smoke-manifest", + "/smoke.json", + "--harness-root", + "/harness", + "--harness-python", + "/harness/python", + "--processor-revision", + "processor-sha", + "--output-dir", + "/output", + "--powercontext-revision", + "pc-sha", + "--integration-revision", + "integration-sha", + "--judge-price-policy", + "0.3 USD per million", + ], + ) + + assert result.exit_code != 0 + assert "--judge-price-policy must be a JSON object" in result.output diff --git a/evaluation/tests/unit/test_longmemeval_v2_costs.py b/evaluation/tests/unit/test_longmemeval_v2_costs.py new file mode 100644 index 0000000000..f632dd8822 --- /dev/null +++ b/evaluation/tests/unit/test_longmemeval_v2_costs.py @@ -0,0 +1,688 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +"""Model-free unit tests for explicit price policies and usage-based cost records.""" + +from __future__ import annotations + +import json +from pathlib import Path +from types import SimpleNamespace +from typing import Any + +import pytest + +from powercontext_eval.benchmarks.longmemeval_v2.catalog import SmokeSelection +from powercontext_eval.benchmarks.longmemeval_v2.costs import ( + CostPolicyError, + ModelPricePolicy, + cost_policy_record, + model_free_cost_block, + parse_cost_policy, + usage_cost_block, + usage_cost_usd, +) + +# DeepSeek's cache-hit input rate is far below its cache-miss rate, so a single blended +# input price would misprice a long, highly repeated prompt. +PRICE_POLICY = { + "provider": "deepseek-openai", + "model": "deepseek-flash", + "currency": "USD", + "input_cache_hit_price_per_million": 0.006, + "input_cache_miss_price_per_million": 0.3, + "output_price_per_million": 1.2, + "price_policy_revision": "deepseek-public-list-2026-09", +} + + +def test_parse_cost_policy_keeps_an_unconfigured_policy_null() -> None: + assert parse_cost_policy(None) is None + + +def test_parse_cost_policy_reads_explicit_prices_model_and_revision() -> None: + policy = parse_cost_policy(PRICE_POLICY) + + assert policy == ModelPricePolicy( + provider="deepseek-openai", + model="deepseek-flash", + currency="USD", + input_cache_hit_price_per_million=0.006, + input_cache_miss_price_per_million=0.3, + output_price_per_million=1.2, + price_policy_revision="deepseek-public-list-2026-09", + ) + assert cost_policy_record(policy) == PRICE_POLICY + + +@pytest.mark.parametrize( + "value", + [ + {"provider": "deepseek-openai"}, + {**PRICE_POLICY, "unexpected": 1}, + {**PRICE_POLICY, "provider": " "}, + {**PRICE_POLICY, "model": ""}, + {**PRICE_POLICY, "currency": ""}, + {**PRICE_POLICY, "price_policy_revision": ""}, + {**PRICE_POLICY, "input_cache_hit_price_per_million": -0.1}, + {**PRICE_POLICY, "input_cache_miss_price_per_million": -1}, + {**PRICE_POLICY, "output_price_per_million": -1}, + {**PRICE_POLICY, "input_cache_hit_price_per_million": True}, + {**PRICE_POLICY, "input_cache_miss_price_per_million": "free"}, + {**PRICE_POLICY, "output_price_per_million": float("inf")}, + {**PRICE_POLICY, "input_cache_hit_price_per_million": float("nan")}, + ], +) +def test_parse_cost_policy_rejects_a_malformed_explicit_policy(value: object) -> None: + with pytest.raises(CostPolicyError): + parse_cost_policy(value) + + +def test_parse_cost_policy_rejects_a_non_object_policy() -> None: + with pytest.raises(CostPolicyError, match="JSON object"): + parse_cost_policy("0.3 USD per million") + + +@pytest.mark.parametrize("currency", ["CNY", "EUR", "usd-cents", "RMB"]) +def test_parse_cost_policy_rejects_a_non_usd_currency(currency: str) -> None: + """The amount fields are named ``*_usd``, so a non-USD policy must be refused.""" + + with pytest.raises(CostPolicyError, match="currency must be USD"): + parse_cost_policy({**PRICE_POLICY, "currency": currency}) + + +def test_parse_cost_policy_accepts_a_lowercase_usd_currency() -> None: + policy = parse_cost_policy({**PRICE_POLICY, "currency": "usd"}) + + assert policy is not None + assert policy.currency == "USD" + + +def test_usage_cost_usd_prices_cache_hit_and_miss_input_separately() -> None: + policy = parse_cost_policy(PRICE_POLICY) + assert policy is not None + + # 1M cached input at 0.006/M, 1M uncached input at 0.3/M, 1M output at 1.2/M. + priced = usage_cost_usd(policy, cache_hit_tokens=1_000_000, cache_miss_tokens=1_000_000, output_tokens=1_000_000) + assert priced == 1.506 + # Pricing every input token at the cache-miss rate would overcharge this by 0.294 USD. + blended = usage_cost_usd(policy, cache_hit_tokens=0, cache_miss_tokens=2_000_000, output_tokens=1_000_000) + assert blended == 1.8 + + +def test_usage_cost_usd_keeps_zero_usage_at_zero_cost() -> None: + policy = parse_cost_policy(PRICE_POLICY) + assert policy is not None + + assert usage_cost_usd(policy, cache_hit_tokens=0, cache_miss_tokens=0, output_tokens=0) == 0.0 + + +def test_usage_cost_usd_rejects_a_negative_or_non_integer_token_count() -> None: + policy = parse_cost_policy(PRICE_POLICY) + assert policy is not None + + with pytest.raises(CostPolicyError, match="non-negative integer"): + usage_cost_usd(policy, cache_hit_tokens=-1, cache_miss_tokens=0, output_tokens=0) + with pytest.raises(CostPolicyError, match="non-negative integer"): + usage_cost_usd(policy, cache_hit_tokens=True, cache_miss_tokens=0, output_tokens=0) # type: ignore[arg-type] + + +def test_usage_cost_block_reports_null_cost_and_the_reason_without_a_policy() -> None: + block = usage_cost_block( + None, + provider="deepseek-openai", + model="deepseek-flash", + input_tokens=123, + cache_hit_tokens=100, + cache_miss_tokens=23, + output_tokens=7, + ) + + assert block["cost_usd"] is None + assert block["currency"] is None + assert block["input_cache_hit_tokens"] == 100 + assert block["input_cache_miss_tokens"] == 23 + assert block["unavailable_reason"] == "no price policy was configured for this run" + + +@pytest.mark.parametrize( + ("provider", "model"), + [ + ("anthropic-compatible", "deepseek-flash"), + ("deepseek-openai", "deepseek-v4-pro"), + ], +) +def test_usage_cost_block_requires_the_policy_to_match_provider_and_model(provider: str, model: str) -> None: + """The same provider may serve several models at different prices, so both must match.""" + + policy = parse_cost_policy(PRICE_POLICY) + assert policy is not None + + block = usage_cost_block( + policy, + provider=provider, + model=model, + input_tokens=10, + cache_hit_tokens=5, + cache_miss_tokens=5, + output_tokens=5, + ) + + assert block["cost_usd"] is None + assert block["input_cache_hit_price_per_million"] is None + assert f"model {model!r}" in str(block["unavailable_reason"]) + + +def test_usage_cost_block_refuses_to_price_usage_without_a_cache_split() -> None: + """A single blended input total cannot be priced when cache-hit and miss rates differ.""" + + policy = parse_cost_policy(PRICE_POLICY) + assert policy is not None + + block = usage_cost_block( + policy, + provider="deepseek-openai", + model="deepseek-flash", + input_tokens=1_000_000, + cache_hit_tokens=None, + cache_miss_tokens=None, + output_tokens=0, + ) + + assert block["cost_usd"] is None + assert "cache hit/miss split" in str(block["unavailable_reason"]) + + +@pytest.mark.parametrize( + ("cache_hit_tokens", "cache_miss_tokens"), + [(100, None), (None, 100)], +) +def test_usage_cost_block_refuses_a_partial_cache_split( + cache_hit_tokens: int | None, cache_miss_tokens: int | None +) -> None: + policy = parse_cost_policy(PRICE_POLICY) + assert policy is not None + + block = usage_cost_block( + policy, + provider="deepseek-openai", + model="deepseek-flash", + input_tokens=100, + cache_hit_tokens=cache_hit_tokens, + cache_miss_tokens=cache_miss_tokens, + output_tokens=0, + ) + + assert block["cost_usd"] is None + assert "only one side" in str(block["unavailable_reason"]) + + +def test_usage_cost_block_refuses_a_cache_split_that_does_not_match_input_total() -> None: + policy = parse_cost_policy(PRICE_POLICY) + assert policy is not None + + block = usage_cost_block( + policy, + provider="deepseek-openai", + model="deepseek-flash", + input_tokens=1_000, + cache_hit_tokens=100, + cache_miss_tokens=200, + output_tokens=0, + ) + + assert block["cost_usd"] is None + assert "do not equal" in str(block["unavailable_reason"]) + + +@pytest.mark.parametrize( + "usage", + [ + {"prompt_tokens": 100, "completion_tokens": 1, "prompt_cache_hit_tokens": 100}, + {"prompt_tokens": 100, "completion_tokens": 1, "prompt_cache_miss_tokens": 100}, + ], +) +def test_deepseek_normalization_drops_a_partial_cache_split(usage: dict[str, int]) -> None: + from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import _normalize_deepseek_response + + response = _normalize_deepseek_response( + { + "model": "deepseek-flash", + "choices": [{"message": {"content": "answer"}, "finish_reason": "stop"}], + "usage": usage, + } + ) + + normalized = response["usage"] + assert isinstance(normalized, dict) + assert "input_cache_hit_tokens" not in normalized + assert "input_cache_miss_tokens" not in normalized + + +def test_usage_cost_block_records_cache_split_cost_and_policy_identity() -> None: + policy = parse_cost_policy(PRICE_POLICY) + assert policy is not None + + block = usage_cost_block( + policy, + provider="deepseek-openai", + model="deepseek-flash", + input_tokens=2_000_000, + cache_hit_tokens=1_000_000, + cache_miss_tokens=1_000_000, + output_tokens=1_000_000, + ) + + assert block["cost_usd"] == 1.506 + assert block["currency"] == "USD" + assert block["model"] == "deepseek-flash" + assert block["input_cache_hit_price_per_million"] == 0.006 + assert block["input_cache_miss_price_per_million"] == 0.3 + assert block["output_price_per_million"] == 1.2 + assert block["price_policy_revision"] == "deepseek-public-list-2026-09" + assert block["unavailable_reason"] is None + + +def test_model_free_cost_block_records_zero_tokens_zero_cost_and_the_reason() -> None: + block = model_free_cost_block(stage="ingestion", reason="the adapter ingests without a model") + + assert block == { + "stage": "ingestion", + "input_tokens": 0, + "input_cache_hit_tokens": 0, + "input_cache_miss_tokens": 0, + "output_tokens": 0, + "cost_usd": 0.0, + "reason": "the adapter ingests without a model", + } + + +def test_cost_policy_record_is_null_without_a_policy() -> None: + assert cost_policy_record(None) is None + + +def _write_json(path: Path, value: object) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(value), encoding="utf-8") + + +def _prepared_artifacts(root: Path) -> Path: + prepared = root / "prepared" + prepared.mkdir() + _write_json( + prepared / "prepare-manifest.json", + {"classification": "smoke-subset-prepare-only", "reader": None, "judge": None}, + ) + _write_json( + prepared / "prepare-summary.json", + {"classification": "smoke-subset-prepare-only", "question_count": 10, "failed": 0}, + ) + with (prepared / "prepared-prompts.jsonl").open("w", encoding="utf-8") as stream: + for index in range(10): + stream.write( + json.dumps( + { + "question_id": f"q-{index}", + "sequence": index + 1, + "messages": [ + {"role": "system", "content": "system"}, + {"role": "user", "content": [{"type": "text", "text": "question"}]}, + ], + } + ) + + "\n" + ) + return prepared + + +class _CountingReader: + """Return normalized DeepSeek-shaped usage, including the cache split, without any network call.""" + + def __init__(self, *, with_cache_split: bool = True) -> None: + self.calls = 0 + self._with_cache_split = with_cache_split + + def complete(self, *, system: str, content: list[dict[str, object]]) -> dict[str, object]: + self.calls += 1 + usage: dict[str, object] = { + "input_tokens": 2_000_000, + "output_tokens": 1_000_000, + } + if self._with_cache_split: + usage["input_cache_hit_tokens"] = 1_000_000 + usage["input_cache_miss_tokens"] = 1_000_000 + return { + "model": "deepseek-flash", + "stop_reason": "end_turn", + "content": [{"type": "text", "text": "\\boxed{answer}"}], + "usage": usage, + } + + +def test_deepseek_response_normalization_keeps_the_native_cache_split() -> None: + """DeepSeek's ``prompt_cache_*_tokens`` fields must survive into normalized usage.""" + + from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import _normalize_deepseek_response + + normalized = _normalize_deepseek_response( + { + "model": "deepseek-flash", + "choices": [{"finish_reason": "stop", "message": {"content": "\\boxed{answer}"}}], + "usage": { + "prompt_tokens": 2_000_000, + "completion_tokens": 1_000_000, + "prompt_cache_hit_tokens": 1_500_000, + "prompt_cache_miss_tokens": 500_000, + }, + } + ) + + assert normalized["usage"] == { + "input_tokens": 2_000_000, + "input_cache_hit_tokens": 1_500_000, + "input_cache_miss_tokens": 500_000, + "output_tokens": 1_000_000, + } + + +def test_deepseek_response_normalization_omits_an_absent_cache_split() -> None: + from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import _normalize_deepseek_response + + normalized = _normalize_deepseek_response( + { + "model": "deepseek-flash", + "choices": [{"finish_reason": "stop", "message": {"content": "\\boxed{answer}"}}], + "usage": {"prompt_tokens": 10, "completion_tokens": 2}, + } + ) + + assert normalized["usage"] == {"input_tokens": 10, "output_tokens": 2} + + +def test_reader_keeps_the_deepseek_cache_split_in_its_outputs_and_summary(tmp_path: Path) -> None: + from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import run_reader_smoke + + result = run_reader_smoke( + prepared_dir=_prepared_artifacts(tmp_path), + output_dir=tmp_path / "reader", + provider="deepseek-openai", + model="deepseek-flash", + max_questions=1, + transport=_CountingReader(), + ) + + [record] = [json.loads(line) for line in result.outputs_path.read_text(encoding="utf-8").splitlines()] + assert record["usage"] == { + "input_tokens": 2_000_000, + "input_cache_hit_tokens": 1_000_000, + "input_cache_miss_tokens": 1_000_000, + "output_tokens": 1_000_000, + } + summary = json.loads(result.summary_path.read_text(encoding="utf-8")) + assert summary["usage"] == { + "input_tokens": 2_000_000, + "input_cache_hit_tokens": 1_000_000, + "input_cache_miss_tokens": 1_000_000, + "output_tokens": 1_000_000, + } + + +def test_reader_summary_prices_cache_hit_and_miss_usage_separately(tmp_path: Path) -> None: + from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import run_reader_smoke + + policy = parse_cost_policy(PRICE_POLICY) + transport = _CountingReader() + result = run_reader_smoke( + prepared_dir=_prepared_artifacts(tmp_path), + output_dir=tmp_path / "reader", + provider="deepseek-openai", + model="deepseek-flash", + max_questions=2, + price_policy=policy, + transport=transport, + ) + + summary = json.loads(result.summary_path.read_text(encoding="utf-8")) + assert transport.calls == 2 + # 2M cached at 0.006/M plus 2M uncached at 0.3/M plus 2M out at 1.2/M. + assert summary["cost"]["cost_usd"] == 3.012 + assert summary["cost"]["currency"] == "USD" + assert summary["cost"]["model"] == "deepseek-flash" + assert summary["cost"]["price_policy_revision"] == "deepseek-public-list-2026-09" + manifest = json.loads(result.manifest_path.read_text(encoding="utf-8")) + assert manifest["cost_policy"] == PRICE_POLICY + + +def test_reader_summary_keeps_cost_null_when_a_response_omits_the_cache_split(tmp_path: Path) -> None: + """One response without the split makes the summed split incomplete, so cost stays null.""" + + from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import run_reader_smoke + + policy = parse_cost_policy(PRICE_POLICY) + result = run_reader_smoke( + prepared_dir=_prepared_artifacts(tmp_path), + output_dir=tmp_path / "reader", + provider="deepseek-openai", + model="deepseek-flash", + max_questions=1, + price_policy=policy, + transport=_CountingReader(with_cache_split=False), + ) + + summary = json.loads(result.summary_path.read_text(encoding="utf-8")) + assert summary["cost"]["cost_usd"] is None + assert "cache hit/miss split" in str(summary["cost"]["unavailable_reason"]) + assert "input_cache_hit_tokens" not in summary["usage"] + + +def test_reader_summary_keeps_cost_null_when_the_policy_prices_another_model(tmp_path: Path) -> None: + from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import run_reader_smoke + + policy = parse_cost_policy({**PRICE_POLICY, "model": "deepseek-v4-pro"}) + result = run_reader_smoke( + prepared_dir=_prepared_artifacts(tmp_path), + output_dir=tmp_path / "reader", + provider="deepseek-openai", + model="deepseek-flash", + max_questions=1, + price_policy=policy, + transport=_CountingReader(), + ) + + summary = json.loads(result.summary_path.read_text(encoding="utf-8")) + assert summary["cost"]["cost_usd"] is None + assert "model 'deepseek-v4-pro'" in str(summary["cost"]["unavailable_reason"]) + + +def test_reader_summary_keeps_cost_null_without_a_policy(tmp_path: Path) -> None: + from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import run_reader_smoke + + result = run_reader_smoke( + prepared_dir=_prepared_artifacts(tmp_path), + output_dir=tmp_path / "reader", + provider="deepseek-openai", + model="deepseek-flash", + max_questions=1, + transport=_CountingReader(), + ) + + summary = json.loads(result.summary_path.read_text(encoding="utf-8")) + assert summary["cost"]["cost_usd"] is None + assert summary["cost"]["unavailable_reason"] == "no price policy was configured for this run" + manifest = json.loads(result.manifest_path.read_text(encoding="utf-8")) + assert manifest["cost_policy"] is None + + +def test_reader_summary_never_writes_zero_for_an_unconfigured_cost(tmp_path: Path) -> None: + from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import run_reader_smoke + + result = run_reader_smoke( + prepared_dir=_prepared_artifacts(tmp_path), + output_dir=tmp_path / "reader", + provider="deepseek-openai", + model="deepseek-flash", + max_questions=1, + transport=_CountingReader(), + ) + + recorded = result.summary_path.read_text(encoding="utf-8") + assert '"cost_usd": 0' not in recorded + assert '"cost_usd": null' in recorded + + +def _score_fixture(root: Path) -> tuple[Path, Path, Path]: + reader = root / "reader" + data = root / "data" + reader.mkdir() + data.mkdir() + _write_json(reader / "reader-manifest.json", {"classification": "smoke-subset-reader-only", "judge": None}) + _write_json( + reader / "reader-summary.json", + {"classification": "smoke-subset-reader-only", "question_count": 10, "failed": 0}, + ) + cases = [] + with ( + (reader / "reader-outputs.jsonl").open("w", encoding="utf-8") as outputs, + (data / "questions.jsonl").open("w", encoding="utf-8") as questions, + ): + for index in range(10): + question_id = f"q-{index}" + evaluator = "llm_abstention_checker" if index < 2 else "exact" + outputs.write(json.dumps({"question_id": question_id, "response_text": f"answer-{index}"}) + "\n") + questions.write( + json.dumps( + { + "id": question_id, + "domain": "web", + "question": f"question {index}", + "answer": f"answer-{index}", + "eval_function": evaluator, + } + ) + + "\n" + ) + cases.append({"question_id": question_id, "ability": "static_state"}) + manifest = root / "smoke.json" + _write_json(manifest, {"schema": "powercontext.longmemeval-v2-smoke.v1", "tier": "small", "cases": cases}) + return reader, data, manifest + + +class _FakeJudge: + """Report usage without a cache split, standing in for a provider that omits it.""" + + def __init__(self) -> None: + self.calls = 0 + + def complete(self, *, system: str, content: list[dict[str, object]]) -> dict[str, object]: + self.calls += 1 + return { + "content": [{"type": "text", "text": '{"label":1}'}], + "usage": {"input_tokens": 1_000_000, "output_tokens": 1_000_000}, + } + + +class _CacheAwareJudge: + def complete(self, *, system: str, content: list[dict[str, object]]) -> dict[str, object]: + return { + "content": [{"type": "text", "text": '{"label":1}'}], + "usage": { + "input_tokens": 1_000_000, + "input_cache_hit_tokens": 1_000_000, + "input_cache_miss_tokens": 0, + "output_tokens": 1_000_000, + }, + } + + +class _FakeMetrics: + def eval_name(self, spec: str) -> str: + return spec + + def extract_boxed_answer(self, text: str) -> str: + return text + + def eval_from_spec(self, spec: str, prediction: str, answer: str) -> bool: + return prediction == answer + + def score_to_bool(self, value: Any) -> bool: + return bool(value) + + def _build_abstention_judge_messages(self, **kwargs: Any) -> list[dict[str, str]]: + return [{"role": "system", "content": "abstention"}, {"role": "user", "content": "question"}] + + def _build_gotchas_judge_messages(self, **kwargs: Any) -> list[dict[str, str]]: + return [{"role": "system", "content": "gotchas"}, {"role": "user", "content": "question"}] + + def _parse_llm_binary_judgement(self, text: str) -> tuple[int, str]: + return 1, "accepted" + + +def _run_score(tmp_path: Path, monkeypatch: pytest.MonkeyPatch, *, policy: Any, judge: Any) -> Any: + from powercontext_eval.benchmarks.longmemeval_v2 import score_smoke + from powercontext_eval.benchmarks.longmemeval_v2.score_smoke import run_score_smoke + + reader, data, manifest = _score_fixture(tmp_path) + monkeypatch.setattr(score_smoke, "validate_harness_checkout", lambda root: None) + monkeypatch.setattr(score_smoke, "_load_metrics", lambda root: _FakeMetrics()) + monkeypatch.setattr(score_smoke, "load_dataset_lock", lambda path: SimpleNamespace(tier="small", file_digests={})) + monkeypatch.setattr( + score_smoke, + "LongMemEvalV2Catalog", + SimpleNamespace( + load=lambda *args, **kwargs: SimpleNamespace( + select_smoke=lambda cases: SmokeSelection("small", tuple(cases)) + ) + ), + ) + result = run_score_smoke( + reader_dir=reader, + data_root=data, + dataset_lock=tmp_path / "dataset-lock.json", + smoke_manifest=manifest, + harness_root=tmp_path / "harness", + output_dir=tmp_path / "score", + judge_price_policy=policy, + judge_transport=judge, + ) + return json.loads(result.summary_path.read_text(encoding="utf-8")) + + +def test_score_summary_keeps_judge_cost_null_when_the_judge_omits_the_cache_split( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """A blended Judge input total must not be priced at one rate.""" + + policy = parse_cost_policy(PRICE_POLICY) + summary = _run_score(tmp_path, monkeypatch, policy=policy, judge=_FakeJudge()) + + assert summary["judge_cost"]["cost_usd"] is None + assert "cache hit/miss split" in str(summary["judge_cost"]["unavailable_reason"]) + + +def test_score_summary_prices_a_judge_that_reports_its_cache_split( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + policy = parse_cost_policy(PRICE_POLICY) + summary = _run_score(tmp_path, monkeypatch, policy=policy, judge=_CacheAwareJudge()) + + # 2M cached input at 0.006/M plus 2M output at 1.2/M. + assert summary["judge_cost"]["cost_usd"] == 2.412 + assert summary["judge_cost"]["price_policy_revision"] == "deepseek-public-list-2026-09" + + +def test_score_summary_keeps_judge_cost_null_without_a_policy(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + summary = _run_score(tmp_path, monkeypatch, policy=None, judge=_FakeJudge()) + + assert summary["judge_cost"]["cost_usd"] is None + assert summary["judge_cost"]["unavailable_reason"] == "no price policy was configured for this run" diff --git a/evaluation/tests/unit/test_longmemeval_v2_harness_bootstrap.py b/evaluation/tests/unit/test_longmemeval_v2_harness_bootstrap.py new file mode 100644 index 0000000000..e8dcebda65 --- /dev/null +++ b/evaluation/tests/unit/test_longmemeval_v2_harness_bootstrap.py @@ -0,0 +1,196 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from __future__ import annotations + +import json +from collections.abc import Mapping +from pathlib import Path + +import pytest + +from powercontext_eval.benchmarks.longmemeval_v2 import harness_bootstrap +from powercontext_eval.benchmarks.longmemeval_v2.adapter import AUDIT_SCHEMA, PowerContextMemory + +_MEMORY_CONTRACT = """ +MEMORY_TYPES = {} + +def register_memory(memory_cls): + MEMORY_TYPES[memory_cls.memory_type] = memory_cls + return memory_cls + +def build_memory(config): + return MEMORY_TYPES[config["memory_type"]](config["memory_params"]) +""" + +_HARNESS = """ +import json +import sys +from pathlib import Path + +from memory_modules.memory import build_memory + +payload = json.loads(Path(sys.argv[1]).read_text(encoding="utf-8")) +memory = build_memory(payload["memory_config"]) +memory.configure_runtime( + query_trace_dir=Path("unused"), + generation_temperature=None, + generation_top_p=None, + cancel_event=None, +) +memory.insert(payload["trajectory"]) +memory.set_query_context(query_invocation_id="invocation-1") +try: + context = memory.query(payload["question"], query_image=payload["query_image"]) + post_query = memory.post_query_hook( + query=payload["question"], + query_image=payload["query_image"], + memory_context=context, + ) +finally: + memory.clear_query_context() +Path(sys.argv[2]).write_text( + json.dumps({"context": context, "post_query": post_query}), + encoding="utf-8", +) +""" + + +class FakePowerContext: + def __init__(self) -> None: + self.captures: list[dict[str, object]] = [] + self.memories: list[dict[str, object]] = [] + self.searches: list[dict[str, object]] = [] + + def capture_content_source(self, payload: Mapping[str, object]) -> Mapping[str, object]: + request = dict(payload) + self.captures.append(request) + return {"source": {"name": "content", "source_id": request["source_id"]}} + + def remember_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + request = dict(payload) + self.memories.append(request) + return { + "entry": { + "citation": { + "memory_ref": {"family": "memory", "artifact_id": "memory", "revision": 1}, + "entry_id": "entry-1", + "entry_version_id": "entry-1-v1", + } + } + } + + def search_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + self.searches.append(dict(payload)) + return { + "hits": [ + { + "text": "The remembered assignment procedure.", + "citation": { + "memory_ref": {"family": "memory", "artifact_id": "memory", "revision": 1}, + "entry_id": "entry-1", + "entry_version_id": "entry-1-v1", + }, + } + ] + } + + def prepare_context(self, payload: Mapping[str, object]) -> Mapping[str, object]: + content = "The remembered assignment procedure." + return { + "schema": "powercontext.prepared-context.v1", + "status": "ready", + "content": content, + "content_bytes": len(content.encode()), + } + + +def test_bootstrap_registers_adapter_and_runs_harness_insert_query_chain( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + harness_root = tmp_path / "LongMemEval-V2" + memory_modules = harness_root / "memory_modules" + evaluation = harness_root / "evaluation" + memory_modules.mkdir(parents=True) + evaluation.mkdir() + (memory_modules / "__init__.py").write_text("", encoding="utf-8") + (memory_modules / "memory.py").write_text(_MEMORY_CONTRACT, encoding="utf-8") + (evaluation / "harness.py").write_text(_HARNESS, encoding="utf-8") + + audit_path = tmp_path / "adapter-audit.jsonl" + input_path = tmp_path / "input.json" + output_path = tmp_path / "output.json" + secret_gold = "GOLD-MUST-NOT-ENTER-ADAPTER" + input_path.write_text( + json.dumps( + { + "memory_config": { + "memory_type": "powercontext", + "memory_params": {"scope_id": "smoke-scope", "audit_path": str(audit_path)}, + }, + "trajectory": { + "id": "trajectory-1", + "domain": "enterprise", + "environment": "workarena", + "goal": "Assign the incident.", + "outcome": "success", + "start_url": "https://example.test", + "states": [ + { + "state_index": 0, + "step": 0, + "url": "https://example.test", + "action": "click('assign')", + "thought": "Use the documented group.", + "accessibility_tree": "Assignment group Network", + "screenshot": "screenshots/0.png", + } + ], + "question_type": "procedure", + "gold_answer": secret_gold, + "judge": {"answer": secret_gold}, + }, + "question": "Which assignment group should be used?", + "query_image": "question.png", + } + ), + encoding="utf-8", + ) + + validated: list[Path] = [] + fake = FakePowerContext() + monkeypatch.setattr(harness_bootstrap, "validate_harness_checkout", lambda root: validated.append(root)) + monkeypatch.setattr(PowerContextMemory, "_http_runtime", lambda self: fake) + + harness_bootstrap.run_pinned_harness(harness_root, [str(input_path), str(output_path)]) + + assert validated == [harness_root.resolve()] + assert len(fake.captures) == 1 + assert len(fake.memories) == 1 + assert fake.searches == [ + {"scope_id": "smoke-scope", "query": "Which assignment group should be used?", "limit": 10, "mode": "auto"} + ] + output = json.loads(output_path.read_text(encoding="utf-8")) + assert output["context"] == [{"type": "text", "value": "The remembered assignment procedure."}] + assert output["post_query"]["query_invocation_id"] == "invocation-1" + assert output["post_query"]["result_count"] == 1 + assert output["post_query"]["citations"][0]["entry_id"] == "entry-1" + audit_text = audit_path.read_text(encoding="utf-8") + audit = [json.loads(line) for line in audit_text.splitlines()] + assert [event["operation"] for event in audit] == ["ingest", "query"] + assert all(event["schema"] == AUDIT_SCHEMA for event in audit) + assert audit[1]["query_invocation_id"] == "invocation-1" + assert secret_gold not in audit_text + assert "question_type" not in audit_text + assert "judge" not in audit_text diff --git a/evaluation/tests/unit/test_longmemeval_v2_prepare_smoke.py b/evaluation/tests/unit/test_longmemeval_v2_prepare_smoke.py new file mode 100644 index 0000000000..4231aed8f5 --- /dev/null +++ b/evaluation/tests/unit/test_longmemeval_v2_prepare_smoke.py @@ -0,0 +1,205 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from __future__ import annotations + +import json +import os +import subprocess +import venv +from pathlib import Path +from typing import Any + +import pytest + +from powercontext_eval.benchmarks.longmemeval_v2 import prepare_smoke +from powercontext_eval.benchmarks.longmemeval_v2.prepare_smoke import PrepareSmokeError, prepare_reader_inputs_smoke + + +def retrieval_artifacts(tmp_path: Path) -> Path: + root = tmp_path / "retrieval" + root.mkdir() + (root / "retrieval-manifest.json").write_text( + json.dumps( + { + "classification": "smoke-subset-retrieval-only", + "runtime": {"reader": None, "judge": None}, + } + ), + encoding="utf-8", + ) + (root / "summary.json").write_text( + json.dumps({"classification": "smoke-subset-retrieval-only", "question_count": 10, "failed": 0}), + encoding="utf-8", + ) + (root / "retrieval-results.jsonl").write_text("{}\n", encoding="utf-8") + return root + + +def test_prepare_smoke_launches_pinned_worker_with_hashed_retrieval_inputs( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + retrieval = retrieval_artifacts(tmp_path) + harness = tmp_path / "harness" + harness.mkdir() + harness_python = tmp_path / "python.exe" + harness_python.write_text("placeholder", encoding="utf-8") + output = tmp_path / "prepared" + calls: list[tuple[list[str], dict[str, Any]]] = [] + + def fake_run(command: list[str], **kwargs: Any) -> subprocess.CompletedProcess[str]: + calls.append((command, kwargs)) + prompts_path = Path(command[command.index("--prompts-path") + 1]) + failures_path = Path(command[command.index("--failures-path") + 1]) + summary_path = Path(command[command.index("--summary-path") + 1]) + prompts_path.write_text("{}\n", encoding="utf-8") + failures_path.write_text("", encoding="utf-8") + summary_path.write_text("{}\n", encoding="utf-8") + return subprocess.CompletedProcess(command, 0, "worker-ok\n", "") + + monkeypatch.setattr(prepare_smoke, "validate_harness_checkout", lambda root: None) + monkeypatch.setattr(prepare_smoke.subprocess, "run", fake_run) + + result = prepare_reader_inputs_smoke( + retrieval_dir=retrieval, + harness_root=harness, + harness_python=harness_python, + output_dir=output, + processor_revision="processor-sha", + processor_model="test/processor", + memory_context_max_tokens=123, + ) + + manifest = json.loads(result.manifest_path.read_text(encoding="utf-8")) + assert manifest["processor"] == {"model": "test/processor", "revision": "processor-sha"} + assert manifest["memory_context_max_tokens"] == 123 + assert manifest["reader"] is None + assert manifest["judge"] is None + assert len(calls) == 1 + command, kwargs = calls[0] + assert command[1:3] == ["-m", "powercontext_eval.benchmarks.longmemeval_v2.prepare_worker"] + assert command[command.index("--processor-revision") + 1] == "processor-sha" + assert kwargs["cwd"] == harness + assert kwargs["env"]["PYTHONNOUSERSITE"] == "1" + source_root = Path(prepare_smoke.__file__).parents[3] + assert Path(kwargs["env"]["PYTHONPATH"].split(os.pathsep, 1)[0]).resolve() == source_root.resolve() + + +def test_prepare_smoke_refuses_to_overwrite_before_validating_harness(tmp_path: Path) -> None: + output = tmp_path / "prepared" + output.mkdir() + + with pytest.raises(PrepareSmokeError, match="Refusing to overwrite"): + prepare_reader_inputs_smoke( + retrieval_dir=tmp_path / "missing", + harness_root=tmp_path / "missing-harness", + harness_python=tmp_path / "missing-python", + output_dir=output, + processor_revision="processor-sha", + ) + + +def test_prepare_smoke_resolves_relative_paths_against_the_caller_cwd( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + retrieval = retrieval_artifacts(tmp_path) + harness = tmp_path / "harness" + harness.mkdir() + harness_python = tmp_path / "python.exe" + harness_python.write_text("placeholder", encoding="utf-8") + monkeypatch.chdir(tmp_path) + calls: list[tuple[list[str], dict[str, Any]]] = [] + + def fake_run(command: list[str], **kwargs: Any) -> subprocess.CompletedProcess[str]: + calls.append((command, kwargs)) + prompts_path = Path(command[command.index("--prompts-path") + 1]) + failures_path = Path(command[command.index("--failures-path") + 1]) + summary_path = Path(command[command.index("--summary-path") + 1]) + prompts_path.write_text("{}\n", encoding="utf-8") + failures_path.write_text("", encoding="utf-8") + summary_path.write_text("{}\n", encoding="utf-8") + return subprocess.CompletedProcess(command, 0, "worker-ok\n", "") + + monkeypatch.setattr(prepare_smoke, "validate_harness_checkout", lambda root: None) + monkeypatch.setattr(prepare_smoke.subprocess, "run", fake_run) + + prepare_reader_inputs_smoke( + retrieval_dir=Path("retrieval"), + harness_root=Path("harness"), + harness_python=Path("python.exe"), + output_dir=Path("prepared"), + processor_revision="processor-sha", + ) + + (command, kwargs) = calls[0] + assert Path(command[0]).is_absolute() + assert Path(command[0]).resolve() == harness_python.resolve() + assert Path(command[command.index("--harness-root") + 1]).resolve() == harness.resolve() + assert ( + Path(command[command.index("--input-path") + 1]).resolve() == (retrieval / "retrieval-results.jsonl").resolve() + ) + assert ( + Path(command[command.index("--prompts-path") + 1]).resolve() + == (tmp_path / "prepared" / "prepared-prompts.jsonl").resolve() + ) + assert Path(kwargs["cwd"]).resolve() == harness.resolve() + + +def test_prepare_smoke_launches_the_requested_virtualenv_interpreter( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + venv_root = tmp_path / "harness-venv" + venv.create(venv_root, with_pip=False) + interpreter_suffix = "Scripts/python.exe" if os.name == "nt" else "bin/python" + retrieval_artifacts(tmp_path) + harness = tmp_path / "harness" + harness.mkdir() + monkeypatch.chdir(tmp_path) + real_run = subprocess.run + launches: list[list[str]] = [] + + def probe_run(command: list[str], **kwargs: Any) -> subprocess.CompletedProcess[str]: + # Launch the constructed interpreter for real: the prepare stage depends on the + # requested virtualenv (whose POSIX python is a symlink), not the base one. + launches.append(list(command)) + probe = real_run( + [command[0], "-c", "import sys; print(sys.prefix)"], + capture_output=True, + text=True, + check=False, + ) + assert probe.returncode == 0, probe.stderr + assert Path(probe.stdout.strip()).resolve() == venv_root.resolve() + prompts_path = Path(command[command.index("--prompts-path") + 1]) + failures_path = Path(command[command.index("--failures-path") + 1]) + summary_path = Path(command[command.index("--summary-path") + 1]) + prompts_path.write_text("{}\n", encoding="utf-8") + failures_path.write_text("", encoding="utf-8") + summary_path.write_text("{}\n", encoding="utf-8") + return subprocess.CompletedProcess(command, 0, probe.stdout, "") + + monkeypatch.setattr(prepare_smoke, "validate_harness_checkout", lambda root: None) + monkeypatch.setattr(prepare_smoke.subprocess, "run", probe_run) + + prepare_reader_inputs_smoke( + retrieval_dir=Path("retrieval"), + harness_root=Path("harness"), + harness_python=Path("harness-venv") / interpreter_suffix, + output_dir=Path("prepared"), + processor_revision="processor-sha", + ) + + [command] = launches + assert Path(command[0]).is_absolute() + assert command[0].replace("\\", "/").endswith(f"harness-venv/{interpreter_suffix}") diff --git a/evaluation/tests/unit/test_longmemeval_v2_prepare_worker.py b/evaluation/tests/unit/test_longmemeval_v2_prepare_worker.py new file mode 100644 index 0000000000..9664cf1ae5 --- /dev/null +++ b/evaluation/tests/unit/test_longmemeval_v2_prepare_worker.py @@ -0,0 +1,84 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from __future__ import annotations + +from typing import Any + +from powercontext_eval.benchmarks.longmemeval_v2.prepare_worker import _prepare_record + + +class FakeHarness: + def validate_memory_context_items(self, memory_context: Any, *, question_id: str) -> list[dict[str, str]]: + assert question_id == "question-1" + assert isinstance(memory_context, list) + return memory_context + + def truncate_memory_context( + self, + memory_context: list[dict[str, str]], + *, + max_tokens: int, + question_id: str, + ) -> tuple[list[dict[str, str]], int, int]: + assert max_tokens == 20 + assert question_id == "question-1" + return memory_context[:1], 30, 10 + + def get_system_prompt(self, domain: str) -> str: + assert domain == "enterprise" + return "system" + + def build_messages( + self, + *, + system_prompt: str, + question_text: str, + image_path: str | None, + memory_context: list[dict[str, str]], + ) -> tuple[list[dict[str, Any]], list[dict[str, Any]]]: + assert system_prompt == "system" + assert question_text == "Where is the evidence?" + assert image_path is None + assert memory_context == [{"type": "text", "value": "memory-a"}] + messages = [{"role": "system", "content": system_prompt}, {"role": "user", "content": question_text}] + return messages, messages + + +def test_prepare_record_uses_harness_budget_and_emits_replayable_prompt() -> None: + record = { + "question_id": "question-1", + "domain": "enterprise", + "question": {"text": "Where is the evidence?", "image": None}, + "scope_id": "scope-1", + "haystack_digest": "digest-1", + "gold_answer": "must-not-be-copied", + "question_type": "must-not-be-copied", + "judge": {"must": "not-be-copied"}, + "memory_context": [ + {"type": "text", "value": "memory-a"}, + {"type": "text", "value": "memory-b"}, + ], + } + + prepared = _prepare_record(FakeHarness(), record, sequence=1, memory_context_max_tokens=20) + + assert prepared["memory_context"] == [{"type": "text", "value": "memory-a"}] + assert prepared["memory_context_original_tokens"] == 30 + assert prepared["memory_context_tokens"] == 10 + assert prepared["memory_context_was_truncated"] is True + assert prepared["memory_context_max_tokens"] == 20 + assert prepared["prompt_sha256"] + assert prepared["messages"] == prepared["prompt_messages"] + assert not {"answer", "gold_answer", "question_type", "judge", "scorer", "eval_function"} & set(prepared) diff --git a/evaluation/tests/unit/test_longmemeval_v2_reader_smoke.py b/evaluation/tests/unit/test_longmemeval_v2_reader_smoke.py new file mode 100644 index 0000000000..0038875ac6 --- /dev/null +++ b/evaluation/tests/unit/test_longmemeval_v2_reader_smoke.py @@ -0,0 +1,263 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from __future__ import annotations + +import json +from collections.abc import Mapping +from pathlib import Path +from typing import Any, Self + +import pytest + +from powercontext_eval.benchmarks.longmemeval_v2 import reader_smoke +from powercontext_eval.benchmarks.longmemeval_v2.costs import ModelPricePolicy +from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import ( + AnthropicCompatibleReader, + DeepSeekOpenAIReader, + ReaderSmokeError, + run_reader_smoke, +) + + +@pytest.mark.parametrize("reader_type", [AnthropicCompatibleReader, DeepSeekOpenAIReader]) +@pytest.mark.parametrize("url", ["https://provider.example/?token=secret", "https://provider.example/#token"]) +def test_reader_rejects_query_and_fragment_urls(reader_type: type[object], url: str) -> None: + with pytest.raises(ReaderSmokeError, match="without credentials"): + reader_type(url, token="token", model="model", max_tokens=1, temperature=0.0, timeout_seconds=1) # type: ignore[operator] + + +class FakeReader: + def __init__(self) -> None: + self.calls: list[tuple[str, list[dict[str, object]]]] = [] + + def complete(self, *, system: str, content: list[dict[str, object]]) -> Mapping[str, object]: + self.calls.append((system, content)) + return { + "model": "deepseek-flash-latest", + "stop_reason": "end_turn", + "content": [{"type": "text", "text": "\\boxed{answer}"}], + "usage": {"input_tokens": 123, "output_tokens": 7}, + } + + +def prepared_artifacts(tmp_path: Path) -> Path: + root = tmp_path / "prepared" + root.mkdir() + (root / "prepare-manifest.json").write_text( + json.dumps({"classification": "smoke-subset-prepare-only", "reader": None, "judge": None}), + encoding="utf-8", + ) + (root / "prepare-summary.json").write_text( + json.dumps({"classification": "smoke-subset-prepare-only", "question_count": 10, "failed": 0}), + encoding="utf-8", + ) + with (root / "prepared-prompts.jsonl").open("w", encoding="utf-8") as stream: + for index in range(10): + stream.write( + json.dumps( + { + "question_id": f"question-{index}", + "sequence": index + 1, + "domain": "web", + "scope_id": "scope-1", + "haystack_digest": "digest-1", + "prompt_sha256": f"prompt-{index}", + "gold_answer": "must-not-be-copied", + "messages": [ + {"role": "system", "content": "system"}, + {"role": "user", "content": [{"type": "text", "text": "question"}]}, + ], + } + ) + + "\n" + ) + return root + + +def test_reader_smoke_writes_responses_and_never_persists_credentials_or_gold( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + transport = FakeReader() + output = tmp_path / "reader" + monkeypatch.setenv("PRIVATE_TOKEN_ENV", "super-secret-token") + + result = run_reader_smoke( + prepared_dir=prepared_artifacts(tmp_path), + output_dir=output, + base_url_env="PRIVATE_BASE_URL_ENV", + token_env="PRIVATE_TOKEN_ENV", + model="deepseek-flash-latest", + max_questions=1, + transport=transport, + ) + + assert transport.calls == [("system", [{"type": "text", "text": "question"}])] + manifest = json.loads(result.manifest_path.read_text(encoding="utf-8")) + assert manifest["reader"]["token_env"] == "PRIVATE_TOKEN_ENV" + assert "super-secret-token" not in json.dumps(manifest) + [output_record] = [json.loads(line) for line in result.outputs_path.read_text(encoding="utf-8").splitlines()] + assert output_record["response_text"] == "\\boxed{answer}" + assert output_record["usage"] == {"input_tokens": 123, "output_tokens": 7} + assert "gold_answer" not in output_record + summary = json.loads(result.summary_path.read_text(encoding="utf-8")) + assert summary["usage"] == {"input_tokens": 123, "output_tokens": 7} + assert summary["cost"]["cost_usd"] is None + assert summary["cost"]["unavailable_reason"] == "no price policy was configured for this run" + + +def test_reader_smoke_refuses_to_overwrite_before_loading_prepared_artifacts(tmp_path: Path) -> None: + output = tmp_path / "reader" + output.mkdir() + + with pytest.raises(ReaderSmokeError, match="Refusing to overwrite"): + run_reader_smoke( + prepared_dir=tmp_path / "missing", + output_dir=output, + transport=FakeReader(), + ) + + +def test_deepseek_reader_uses_openai_messages_and_disables_thinking(monkeypatch: pytest.MonkeyPatch) -> None: + requests: list[Any] = [] + + class Response: + status = 200 + + def __enter__(self) -> Self: + return self + + def __exit__(self, *args: object) -> None: + return None + + def read(self) -> bytes: + return json.dumps( + { + "model": "deepseek-flash", + "choices": [{"finish_reason": "stop", "message": {"content": "\\boxed{answer}"}}], + "usage": {"prompt_tokens": 12, "completion_tokens": 3}, + } + ).encode() + + def fake_urlopen(request: Any, *, timeout: float) -> Response: + assert timeout == 30 + requests.append(request) + return Response() + + monkeypatch.setattr(reader_smoke, "urlopen", fake_urlopen) + reader = DeepSeekOpenAIReader( + "https://api.deepseek.com", + token="super-secret-token", + model="deepseek-flash", + max_tokens=512, + temperature=0.0, + timeout_seconds=30, + ) + + response = reader.complete(system="system", content=[{"type": "text", "text": "question"}]) + + assert response["usage"] == {"input_tokens": 12, "output_tokens": 3} + assert response["content"] == [{"type": "text", "text": "\\boxed{answer}"}] + request = requests[0] + assert request.full_url == "https://api.deepseek.com/chat/completions" + payload = json.loads(request.data) + assert payload["model"] == "deepseek-flash" + assert payload["thinking"] == {"type": "disabled"} + assert payload["messages"] == [ + {"role": "system", "content": "system"}, + {"role": "user", "content": [{"type": "text", "text": "question"}]}, + ] + assert "super-secret-token" not in request.data.decode() + + +def test_reader_smoke_preserves_usage_when_a_completed_response_has_no_text( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + class NoTextResponse: + status = 200 + + def __enter__(self) -> Self: + return self + + def __exit__(self, *args: object) -> None: + return None + + def read(self) -> bytes: + return json.dumps( + { + "model": "deepseek-flash", + "choices": [{"finish_reason": "stop", "message": {"content": " "}}], + "usage": { + "prompt_tokens": 100, + "completion_tokens": 20, + "prompt_cache_hit_tokens": 40, + "prompt_cache_miss_tokens": 60, + }, + } + ).encode() + + def fake_urlopen(request: Any, *, timeout: float) -> NoTextResponse: + return NoTextResponse() + + monkeypatch.setattr(reader_smoke, "urlopen", fake_urlopen) + reader = DeepSeekOpenAIReader( + "https://api.deepseek.com", + token="token", + model="deepseek-flash", + max_tokens=512, + temperature=0.0, + timeout_seconds=30, + ) + policy = ModelPricePolicy( + provider="deepseek-openai", + model="deepseek-flash", + currency="USD", + input_cache_hit_price_per_million=1.0, + input_cache_miss_price_per_million=2.0, + output_price_per_million=3.0, + price_policy_revision="test-prices", + ) + output = tmp_path / "reader" + + with pytest.raises(ReaderSmokeError, match="Reader failed for 10"): + run_reader_smoke( + prepared_dir=prepared_artifacts(tmp_path), + output_dir=output, + provider="deepseek-openai", + model="deepseek-flash", + transport=reader, + price_policy=policy, + ) + + summary = json.loads((output / "reader-summary.json").read_text(encoding="utf-8")) + assert summary["failed"] == 10 + assert summary["usage"] == { + "input_tokens": 1_000, + "output_tokens": 200, + "input_cache_hit_tokens": 400, + "input_cache_miss_tokens": 600, + } + assert summary["cost"]["cost_usd"] == pytest.approx(2_200 / 1_000_000) + failures = [ + json.loads(line) for line in (output / "reader-failures.jsonl").read_text(encoding="utf-8").splitlines() + ] + assert len(failures) == 10 + assert all(failure["error_type"] == "ReaderResponseError" for failure in failures) + assert all( + failure["usage"] + == {"input_tokens": 100, "output_tokens": 20, "input_cache_hit_tokens": 40, "input_cache_miss_tokens": 60} + for failure in failures + ) + assert all(isinstance(failure["reader_latency_ms"], (int, float)) for failure in failures) + assert (output / "reader-outputs.jsonl").read_text(encoding="utf-8") == "" diff --git a/evaluation/tests/unit/test_longmemeval_v2_replay_score.py b/evaluation/tests/unit/test_longmemeval_v2_replay_score.py new file mode 100644 index 0000000000..3329dc6398 --- /dev/null +++ b/evaluation/tests/unit/test_longmemeval_v2_replay_score.py @@ -0,0 +1,98 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from __future__ import annotations + +import json +from pathlib import Path +from typing import Any + +import pytest + +from powercontext_eval.benchmarks.longmemeval_v2 import replay_score +from powercontext_eval.benchmarks.longmemeval_v2.replay_score import ReplayScoreError, replay_score_smoke + + +class FakeMetrics: + def eval_name(self, spec: str) -> str: + return spec + + def eval_from_spec(self, spec: str, prediction: str, answer: str) -> bool: + return spec == "exact" and prediction == answer + + def score_to_bool(self, value: Any) -> bool: + return bool(value) + + +def score_artifacts(tmp_path: Path) -> Path: + root = tmp_path / "score" + root.mkdir() + (root / "score-manifest.json").write_text( + json.dumps({"classification": "smoke-subset-score-only"}), encoding="utf-8" + ) + (root / "score-summary.json").write_text(json.dumps({"failed": 0}), encoding="utf-8") + (root / "scoring-inputs.local.jsonl").write_text( + "\n".join( + [ + json.dumps( + { + "question_id": "deterministic", + "eval_function": "exact", + "reference_answer": "answer", + "parsed_prediction": "answer", + "reader_response_sha256": "reader-1", + } + ), + json.dumps( + { + "question_id": "judge", + "eval_function": "llm_gotchas_checker", + "reference_answer": "gold", + "parsed_prediction": "prediction", + "reader_response_sha256": "reader-2", + } + ), + ] + ) + + "\n", + encoding="utf-8", + ) + (root / "judge-outputs.jsonl").write_text(json.dumps({"question_id": "judge", "label": 1}) + "\n", encoding="utf-8") + return root + + +def test_replay_score_uses_saved_judge_label_without_network(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None: + score_dir = score_artifacts(tmp_path) + output = tmp_path / "replay" + monkeypatch.setattr(replay_score, "validate_harness_checkout", lambda root: None) + monkeypatch.setattr(replay_score, "_load_metrics", lambda root: FakeMetrics()) + + result = replay_score_smoke(score_dir=score_dir, harness_root=tmp_path / "harness", output_dir=output) + + summary = json.loads(result.summary_path.read_text(encoding="utf-8")) + assert summary["correct"] == 2 + assert summary["accuracy"] == 1.0 + assert summary["reader_calls"] == 0 + assert summary["judge_calls"] == 0 + results = [json.loads(line) for line in result.results_path.read_text(encoding="utf-8").splitlines()] + assert [row["score_mode"] for row in results] == ["deterministic_replay", "llm_judge_replay"] + assert all("reference_answer" not in row for row in results) + + +def test_replay_score_refuses_to_overwrite_before_loading_inputs(tmp_path: Path) -> None: + output = tmp_path / "replay" + output.mkdir() + + with pytest.raises(ReplayScoreError, match="Refusing to overwrite"): + replay_score_smoke(score_dir=tmp_path / "missing", harness_root=tmp_path / "harness", output_dir=output) diff --git a/evaluation/tests/unit/test_longmemeval_v2_report.py b/evaluation/tests/unit/test_longmemeval_v2_report.py new file mode 100644 index 0000000000..5f32b363ae --- /dev/null +++ b/evaluation/tests/unit/test_longmemeval_v2_report.py @@ -0,0 +1,644 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import json +import re +import shutil +from pathlib import Path +from typing import Any + +import pytest + +from powercontext_eval.benchmarks.longmemeval_v2.report import REPORT_BANNER, ReportError, build_report + +GOLD_SENTINEL = "the gold reference answer that must never reach a report" +SECRET_PATTERNS = (re.compile(r"sk-[A-Za-z0-9_-]{8,}"), re.compile(r"eyJ[A-Za-z0-9_-]{10,}")) +FORBIDDEN_KEYS = frozenset( + {"reference_answer", "gold", "gold_answer", "api_key", "auth_token", "token", "token_env", "password", "secret"} +) + + +def _write_json(path: Path, value: object) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n", encoding="utf-8") + + +def _write_jsonl(path: Path, rows: list[dict[str, object]]) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + with path.open("w", encoding="utf-8", newline="") as stream: + for row in rows: + stream.write(json.dumps(row) + "\n") + + +def _successful_run(root: Path) -> Path: + run = root / "run" + _write_json( + run / "run-manifest.json", + { + "schema": "powercontext.longmemeval-v2-smoke-run.v1", + "classification": "smoke-subset", + "run_id": "smoke-v1", + "experiment_arm": { + "id": "current-memory-hybrid-v1", + "search_mode": "hybrid", + "memory_projection": "deterministic-compact-v1", + "query_projection": "question-text-v1", + "temporal_filter": None, + "task_lens": None, + }, + }, + ) + _write_json( + run / "run-summary.json", + { + "schema": "powercontext.longmemeval-v2-smoke-run-summary.v1", + "classification": "smoke-subset", + "run_id": "smoke-v1", + "status": "completed", + "question_count": 10, + "accuracy": {"correct": 4, "incorrect": 6, "failed": 0, "value": 0.4}, + "artifacts": {"retrieval": "02-retrieval"}, + }, + ) + _write_jsonl( + run / "02-retrieval" / "retrieval-results.jsonl", + [ + { + "question_id": f"q{index}", + "status": "succeeded", + "memory_context": [{"type": "text", "value": "x"}] * 3, + "timings_ms": {"search": 1.0, "format": 0.5, "total": 2.0}, + } + for index in range(10) + ], + ) + _write_jsonl( + run / "02-retrieval" / "adapter-audit.jsonl", + [{"operation": "ingest", "timings_ms": {"total": 4.0}}, {"operation": "query", "timings_ms": {"total": 9.0}}], + ) + _write_json( + run / "02-retrieval" / "summary.json", + { + "question_count": 10, + "failed": 0, + "context_bytes": 900, + "citation_count": 12, + "elapsed_ms": 100.0, + }, + ) + _write_json( + run / "03-prepare" / "prepare-summary.json", + {"question_count": 10, "failed": 0, "memory_context_tokens": 1234, "elapsed_ms": 7.0}, + ) + _write_jsonl(run / "04-reader" / "reader-outputs.jsonl", [{"question_id": "q0", "reader_latency_ms": 5.0}] * 10) + _write_json( + run / "04-reader" / "reader-summary.json", + { + "question_count": 10, + "failed": 0, + "usage": {"input_tokens": 1000, "output_tokens": 200}, + "elapsed_ms": 50.0, + }, + ) + _write_jsonl( + run / "05-score" / "judge-outputs.jsonl", + [ + { + "question_id": "q0", + "evaluator": "llm_abstention_checker", + "label": 1, + "judge_latency_ms": 3.0, + }, + { + "question_id": "q1", + "evaluator": "llm_abstention_checker", + "label": 0, + "judge_latency_ms": 4.0, + }, + {"question_id": "q2", "evaluator": "llm_gotchas_checker", "label": 1, "judge_latency_ms": 5.0}, + ], + ) + _write_json( + run / "05-score" / "score-summary.json", + { + "question_count": 10, + "correct": 4, + "incorrect": 6, + "failed": 0, + "accuracy": 0.4, + "judge_usage": {"input_tokens": 300, "output_tokens": 60}, + "elapsed_ms": 40.0, + }, + ) + _write_json( + run / "06-replay" / "replay-summary.json", + {"question_count": 10, "correct": 4, "incorrect": 6, "failed": 0, "accuracy": 0.4}, + ) + # The score stage keeps local reference answers for replay. A report must never read or copy them. + _write_jsonl( + run / "05-score" / "scoring-inputs.local.jsonl", + [{"question_id": "q0", "reference_answer": GOLD_SENTINEL, "api_key": "sk-abcdefghijklmnop"}], + ) + return run + + +def test_report_summarizes_a_successful_run_from_saved_artifacts(tmp_path: Path) -> None: + run = _successful_run(tmp_path) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["schema"] == "powercontext.longmemeval-v2-smoke-report.v1" + assert report["classification"] == "smoke-subset" + assert report["status"] == "completed" + assert report["experiment_arm"] == { + "id": "current-memory-hybrid-v1", + "search_mode": "hybrid", + "memory_projection": "deterministic-compact-v1", + "query_projection": "question-text-v1", + "temporal_filter": None, + "task_lens": None, + } + assert report["question_count"] == 10 + assert report["accuracy"] == {"correct": 4, "incorrect": 6, "failed": 0, "value": 0.4} + assert report["latency_ms"] == { + "ingest": 4.0, + "retrieval": 20.0, + "prepare": 7.0, + "reader": 50.0, + "judge": 12.0, + } + assert report["context"] == {"items": 30, "bytes": 900, "tokens": 1234, "citations_available": 12} + assert report["usage"] == { + "ingestion_tokens": 0, + "ingestion_tokens_note": ( + "the PowerContext Memory adapter ingests without a model, so no provider reports ingestion usage" + ), + "ingestion_cost": { + "stage": "ingestion", + "input_tokens": 0, + "input_cache_hit_tokens": 0, + "input_cache_miss_tokens": 0, + "output_tokens": 0, + "cost_usd": 0.0, + "reason": ( + "the PowerContext Memory adapter ingests without a model, so no provider reports " + "ingestion usage and ingestion cost is zero" + ), + }, + "reader_input_tokens": 1000, + "reader_output_tokens": 200, + "judge_input_tokens": 300, + "judge_output_tokens": 60, + "reader_cost": None, + "judge_cost": None, + "estimated_cost_usd": None, + "estimated_cost_note": ( + "the Reader stage ran but its summary recorded no cost block; " + "the Judge made 1 model call(s) but recorded no priced cost" + ), + } + assert report["abstention"] == {"count": 2, "correct": 1, "incorrect": 1, "unavailable": False} + assert report["failures"] == { + "configuration": 0, + "infrastructure": 0, + "retrieval": 0, + "generation": 0, + "judge": 0, + "integrity": 0, + } + assert report["artifacts"]["retrieval"] == "02-retrieval" + assert report["artifacts"]["score"] == "05-score" + + markdown = result.markdown_path.read_text(encoding="utf-8") + assert markdown.splitlines()[0] == REPORT_BANNER + assert "Accuracy: 4/10 (0.4)" in markdown + assert "Arm: current-memory-hybrid-v1" in markdown + assert "not a complete benchmark result" in markdown + assert "evaluation/docs/longmemeval-v2-full-run.md" in markdown + + +def test_report_uses_relative_latency_from_saved_artifacts_only(tmp_path: Path) -> None: + run = _successful_run(tmp_path) + result = build_report(run_dir=run) + report = json.loads(result.report_path.read_text(encoding="utf-8")) + + assert report["latency_ms"]["ingest"] == 4.0 # only the ingest audit row, never the query row + assert report["latency_ms"]["retrieval"] == 20.0 # ten saved queries at 2 ms each + + +def test_report_counts_failures_by_class_from_run_and_stage_artifacts(tmp_path: Path) -> None: + run = _successful_run(tmp_path) + _write_jsonl( + run / "failures.jsonl", + [ + {"phase": "reader", "error_class": "configuration", "summary": "missing key"}, + {"phase": "reader", "error_class": "generation", "summary": "http 500"}, + ], + ) + _write_jsonl(run / "04-reader" / "reader-failures.jsonl", [{"question_id": "q1", "phase": "reader"}]) + _write_jsonl( + run / "02-retrieval" / "failures.jsonl", + [ + {"question_id": "q2", "phase": "query", "category": "integration"}, + {"question_id": "q3", "phase": "query", "category": "infrastructure"}, + ], + ) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["failures"] == { + "configuration": 1, + "infrastructure": 1, + "retrieval": 1, + "generation": 2, + "judge": 0, + "integrity": 0, + } + + +PRICE_POLICY_RECORD = { + "provider": "deepseek-openai", + "model": "deepseek-flash", + "currency": "USD", + "input_cache_hit_price_per_million": 0.006, + "input_cache_miss_price_per_million": 0.3, + "output_price_per_million": 1.2, + "price_policy_revision": "deepseek-public-list-2026-09", +} + + +def _stage_cost_record(*, cost_usd: float | None, model: str = "deepseek-flash") -> dict[str, object]: + priced = cost_usd is not None + return { + "provider": "deepseek-openai", + "model": model, + "input_tokens": 1300, + "input_cache_hit_tokens": 1000, + "input_cache_miss_tokens": 300, + "output_tokens": 260, + "cost_usd": cost_usd, + "currency": "USD" if priced else None, + "input_cache_hit_price_per_million": 0.006 if priced else None, + "input_cache_miss_price_per_million": 0.3 if priced else None, + "output_price_per_million": 1.2 if priced else None, + "price_policy_revision": "deepseek-public-list-2026-09" if priced else None, + "unavailable_reason": None if priced else "no price policy was configured for this run", + } + + +def _costed_run(tmp_path: Path, *, reader_cost: float | None, judge_cost: float | None) -> Path: + """Build a run whose manifest says both model stages ran and record stage costs.""" + + run = _successful_run(tmp_path) + manifest = json.loads((run / "run-manifest.json").read_text(encoding="utf-8")) + manifest["modes"] = {"reader": True, "score": True} + manifest["reader"] = {"provider": "deepseek-openai", "model": "deepseek-flash"} + manifest["judge"] = {"provider": "deepseek-openai", "model": "deepseek-flash"} + manifest["cost_policy"] = {"reader": PRICE_POLICY_RECORD, "judge": PRICE_POLICY_RECORD} + _write_json(run / "run-manifest.json", manifest) + reader_summary = json.loads((run / "04-reader" / "reader-summary.json").read_text(encoding="utf-8")) + reader_summary["cost"] = _stage_cost_record(cost_usd=reader_cost) + _write_json(run / "04-reader" / "reader-summary.json", reader_summary) + score_summary = json.loads((run / "05-score" / "score-summary.json").read_text(encoding="utf-8")) + score_summary["judge_calls"] = 3 + score_summary["judge_cost"] = _stage_cost_record(cost_usd=judge_cost) + _write_json(run / "05-score" / "score-summary.json", score_summary) + return run + + +def test_report_sums_stage_costs_recorded_under_one_price_policy(tmp_path: Path) -> None: + run = _costed_run(tmp_path, reader_cost=0.00049, judge_cost=0.000147) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["usage"]["reader_cost"]["cost_usd"] == 0.00049 + assert report["usage"]["judge_cost"]["cost_usd"] == 0.000147 + assert report["usage"]["estimated_cost_usd"] == 0.000637 + assert report["usage"]["estimated_cost_note"] == ( + "sum of the Reader and Judge usage costs recorded under the run's explicit price policy" + ) + markdown = result.markdown_path.read_text(encoding="utf-8") + assert "Cost: " in markdown + assert "total_usd=0.000637" in markdown + + +def test_report_keeps_the_total_null_when_an_executed_judge_cost_block_is_missing(tmp_path: Path) -> None: + """A Judge that ran without a recorded cost block must not be silently excluded.""" + + run = _costed_run(tmp_path, reader_cost=0.00049, judge_cost=0.000147) + score_summary = json.loads((run / "05-score" / "score-summary.json").read_text(encoding="utf-8")) + del score_summary["judge_cost"] + _write_json(run / "05-score" / "score-summary.json", score_summary) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["usage"]["reader_cost"]["cost_usd"] == 0.00049 + assert report["usage"]["estimated_cost_usd"] is None + assert "Judge made 3 model call(s) but recorded no priced cost" in report["usage"]["estimated_cost_note"] + + +def test_report_keeps_the_total_null_when_an_executed_reader_cost_block_is_missing(tmp_path: Path) -> None: + run = _costed_run(tmp_path, reader_cost=0.00049, judge_cost=0.000147) + reader_summary = json.loads((run / "04-reader" / "reader-summary.json").read_text(encoding="utf-8")) + del reader_summary["cost"] + _write_json(run / "04-reader" / "reader-summary.json", reader_summary) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["usage"]["estimated_cost_usd"] is None + assert "Reader stage ran but its summary recorded no cost block" in report["usage"]["estimated_cost_note"] + + +def test_report_keeps_the_total_null_when_a_configured_reader_summary_is_absent(tmp_path: Path) -> None: + """A manifest that configured the Reader but kept no summary must not be totalled.""" + + run = _costed_run(tmp_path, reader_cost=0.00049, judge_cost=0.000147) + shutil.rmtree(run / "04-reader") + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["usage"]["estimated_cost_usd"] is None + assert ( + "configured to run the Reader but no Reader summary recorded a cost" in report["usage"]["estimated_cost_note"] + ) + + +def test_report_keeps_the_total_null_when_an_executed_stage_was_unpriced(tmp_path: Path) -> None: + run = _costed_run(tmp_path, reader_cost=None, judge_cost=0.000147) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["usage"]["estimated_cost_usd"] is None + assert report["usage"]["reader_cost"]["cost_usd"] is None + assert report["usage"]["reader_cost"]["unavailable_reason"] == "no price policy was configured for this run" + assert "the Reader recorded non-zero usage without a priced cost" in report["usage"]["estimated_cost_note"] + + +def test_report_keeps_the_total_null_when_stage_costs_use_different_policy_revisions(tmp_path: Path) -> None: + run = _costed_run(tmp_path, reader_cost=0.00049, judge_cost=0.000147) + score_summary = json.loads((run / "05-score" / "score-summary.json").read_text(encoding="utf-8")) + score_summary["judge_cost"]["price_policy_revision"] = "deepseek-public-list-2026-10" + _write_json(run / "05-score" / "score-summary.json", score_summary) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["usage"]["estimated_cost_usd"] is None + assert "configured price policy field price_policy_revision" in report["usage"]["estimated_cost_note"] + + +def test_report_keeps_the_total_null_when_the_recorded_model_is_not_the_configured_model(tmp_path: Path) -> None: + run = _costed_run(tmp_path, reader_cost=0.00049, judge_cost=0.000147) + score_summary = json.loads((run / "05-score" / "score-summary.json").read_text(encoding="utf-8")) + score_summary["judge_cost"]["model"] = "deepseek-v4-pro" + _write_json(run / "05-score" / "score-summary.json", score_summary) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["usage"]["estimated_cost_usd"] is None + assert "not the configured 'deepseek-flash'" in report["usage"]["estimated_cost_note"] + + +def test_report_totals_a_judge_that_made_no_model_call_as_zero_judge_cost(tmp_path: Path) -> None: + """A score stage with no LLM-judge questions owes no Judge cost and must not block the total.""" + + run = _costed_run(tmp_path, reader_cost=0.00049, judge_cost=None) + score_summary = json.loads((run / "05-score" / "score-summary.json").read_text(encoding="utf-8")) + score_summary["judge_calls"] = 0 + score_summary["judge_usage"] = {"input_tokens": 0, "output_tokens": 0} + score_summary["judge_cost"] = { + **_stage_cost_record(cost_usd=None), + "input_tokens": 0, + "input_cache_hit_tokens": 0, + "input_cache_miss_tokens": 0, + "output_tokens": 0, + "unavailable_reason": "no price policy was configured for this run", + } + _write_json(run / "05-score" / "score-summary.json", score_summary) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["usage"]["estimated_cost_usd"] == 0.00049 + assert report["usage"]["estimated_cost_note"] == ( + "sum of the Reader and Judge usage costs recorded under the run's explicit price policy" + ) + + +def test_report_ignores_a_zero_call_zero_usage_judge_cost_when_its_policy_differs(tmp_path: Path) -> None: + run = _costed_run(tmp_path, reader_cost=0.00049, judge_cost=0.0) + score_summary = json.loads((run / "05-score" / "score-summary.json").read_text(encoding="utf-8")) + score_summary["judge_calls"] = 0 + score_summary["judge_usage"] = {"input_tokens": 0, "output_tokens": 0} + score_summary["judge_cost"].update( + { + "input_tokens": 0, + "input_cache_hit_tokens": 0, + "input_cache_miss_tokens": 0, + "output_tokens": 0, + "price_policy_revision": "unused-judge-policy", + } + ) + _write_json(run / "05-score" / "score-summary.json", score_summary) + + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["usage"]["estimated_cost_usd"] == 0.00049 + + +def test_report_rejects_a_zero_call_judge_with_a_priced_nonzero_usage(tmp_path: Path) -> None: + run = _costed_run(tmp_path, reader_cost=0.00049, judge_cost=0.000147) + score_summary = json.loads((run / "05-score" / "score-summary.json").read_text(encoding="utf-8")) + score_summary["judge_calls"] = 0 + _write_json(run / "05-score" / "score-summary.json", score_summary) + + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["usage"]["estimated_cost_usd"] is None + assert "despite zero recorded model calls" in report["usage"]["estimated_cost_note"] + + +def test_report_rejects_a_priced_cost_that_does_not_match_its_configured_provider(tmp_path: Path) -> None: + run = _costed_run(tmp_path, reader_cost=0.00049, judge_cost=0.000147) + reader_summary = json.loads((run / "04-reader" / "reader-summary.json").read_text(encoding="utf-8")) + reader_summary["cost"]["provider"] = "anthropic-compatible" + _write_json(run / "04-reader" / "reader-summary.json", reader_summary) + + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["usage"]["estimated_cost_usd"] is None + assert "not the configured 'deepseek-openai'" in report["usage"]["estimated_cost_note"] + + +def test_report_keeps_the_total_null_when_a_judge_without_calls_still_recorded_usage(tmp_path: Path) -> None: + """A block claiming no Judge call but non-zero usage is self-contradictory, so it blocks the total.""" + + run = _costed_run(tmp_path, reader_cost=0.00049, judge_cost=None) + score_summary = json.loads((run / "05-score" / "score-summary.json").read_text(encoding="utf-8")) + score_summary["judge_calls"] = 0 + score_summary["judge_cost"] = _stage_cost_record(cost_usd=None) + _write_json(run / "05-score" / "score-summary.json", score_summary) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["usage"]["estimated_cost_usd"] is None + assert "despite zero recorded model calls" in report["usage"]["estimated_cost_note"] + + +def test_report_keeps_the_total_null_without_a_price_policy(tmp_path: Path) -> None: + run = _successful_run(tmp_path) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["usage"]["estimated_cost_usd"] is None + # The fixture ran both model stages, so both missing priced costs must be named. + note = report["usage"]["estimated_cost_note"] + assert "the Reader stage ran but its summary recorded no cost block" in note + assert "the Judge made 1 model call(s) but recorded no priced cost" in note + + +def test_report_covers_a_run_that_stopped_after_a_failed_phase(tmp_path: Path) -> None: + run = _successful_run(tmp_path) + for name in ("04-reader", "05-score", "06-replay"): + shutil.rmtree(run / name) + _write_json( + run / "run-summary.json", + { + "classification": "smoke-subset", + "status": "failed", + "question_count": 10, + "accuracy": None, + "completed_phases": ["preflight", "retrieval", "prepare"], + "failed_phase": "prepare", + }, + ) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["status"] == "failed" + assert report["accuracy"] is None + assert report["usage"]["reader_input_tokens"] is None + assert report["latency_ms"]["reader"] is None + assert report["abstention"] == {"count": 0, "correct": None, "incorrect": None, "unavailable": True} + assert report["artifacts"]["score"] is None + markdown = result.markdown_path.read_text(encoding="utf-8") + assert "Accuracy: unavailable" in markdown + assert markdown.splitlines()[0] == REPORT_BANNER + + +def test_report_covers_a_model_free_run_without_inventing_accuracy(tmp_path: Path) -> None: + run = tmp_path / "model-free" + _write_json(run / "run-manifest.json", {"classification": "smoke-subset", "run_id": "model-free"}) + _write_json( + run / "run-summary.json", + {"classification": "smoke-subset", "status": "partial", "question_count": 10, "accuracy": None}, + ) + _write_json(run / "02-retrieval" / "summary.json", {"question_count": 10, "failed": 0, "context_bytes": 10}) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["status"] == "partial" + assert report["accuracy"] is None + assert report["question_count"] == 10 + assert report["usage"]["estimated_cost_usd"] is None + assert report["latency_ms"]["judge"] is None + + +def test_report_never_copies_reference_answers_or_credentials(tmp_path: Path) -> None: + run = _successful_run(tmp_path) + result = build_report(run_dir=run) + + for path in (result.report_path, result.markdown_path): + text = path.read_text(encoding="utf-8") + assert GOLD_SENTINEL not in text + assert "sk-abcdefghijklmnop" not in text + for pattern in SECRET_PATTERNS: + assert pattern.search(text) is None + report = json.loads(result.report_path.read_text(encoding="utf-8")) + for key, value in _walk(report): + assert key.rsplit(".", 1)[-1].lower() not in FORBIDDEN_KEYS, key + assert GOLD_SENTINEL not in str(value) + + +def test_report_refuses_to_overwrite_an_existing_report(tmp_path: Path) -> None: + run = _successful_run(tmp_path) + build_report(run_dir=run) + + with pytest.raises(ReportError, match="Cannot write report artifact"): + build_report(run_dir=run) + + +def test_report_can_be_written_to_a_new_separate_directory(tmp_path: Path) -> None: + run = _successful_run(tmp_path) + target = tmp_path / "report" + result = build_report(run_dir=run, output_dir=target) + + assert result.report_path == target / "report.json" + assert (target / "report.md").is_file() + assert not (run / "report.json").exists() + + with pytest.raises(ReportError, match="Refusing to overwrite report artifacts"): + build_report(run_dir=run, output_dir=target) + + +def test_report_rejects_a_missing_run_directory(tmp_path: Path) -> None: + with pytest.raises(ReportError, match="does not exist"): + build_report(run_dir=tmp_path / "absent") + + +def test_report_of_a_run_without_an_arm_records_no_arm(tmp_path: Path) -> None: + run = tmp_path / "legacy-run" + _write_json(run / "run-manifest.json", {"classification": "smoke-subset", "run_id": "legacy"}) + _write_json( + run / "run-summary.json", + {"classification": "smoke-subset", "status": "partial", "question_count": 10, "accuracy": None}, + ) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["experiment_arm"] is None + markdown = result.markdown_path.read_text(encoding="utf-8") + assert "Arm: unavailable" in markdown + + +def test_report_counts_hybrid_mode_mismatch_as_integrity_failures(tmp_path: Path) -> None: + run = _successful_run(tmp_path) + _write_jsonl( + run / "02-retrieval" / "failures.jsonl", + [ + {"question_id": "q0", "phase": "query", "category": "integrity"}, + {"question_id": "q1", "phase": "query", "category": "integration"}, + {"question_id": "q2", "phase": "ingest", "category": "infrastructure"}, + ], + ) + result = build_report(run_dir=run) + + report = json.loads(result.report_path.read_text(encoding="utf-8")) + assert report["failures"]["integrity"] == 1 + assert report["failures"]["retrieval"] == 1 + assert report["failures"]["infrastructure"] == 1 + + +def _walk(value: Any, prefix: str = "") -> list[tuple[str, object]]: + found: list[tuple[str, object]] = [] + if isinstance(value, dict): + for key, item in value.items(): + found.extend(_walk(item, f"{prefix}.{key}")) + elif isinstance(value, list): + for item in value: + found.extend(_walk(item, prefix)) + else: + found.append((prefix, value)) + return found diff --git a/evaluation/tests/unit/test_longmemeval_v2_retrieval_smoke.py b/evaluation/tests/unit/test_longmemeval_v2_retrieval_smoke.py new file mode 100644 index 0000000000..024170ede1 --- /dev/null +++ b/evaluation/tests/unit/test_longmemeval_v2_retrieval_smoke.py @@ -0,0 +1,574 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from __future__ import annotations + +import hashlib +import json +from collections.abc import Mapping +from pathlib import Path +from types import SimpleNamespace +from typing import Any + +import pytest + +from powercontext_eval.benchmarks.longmemeval_v2 import retrieval_smoke +from powercontext_eval.benchmarks.longmemeval_v2.arms import ExperimentArmError +from powercontext_eval.benchmarks.longmemeval_v2.retrieval_smoke import ( + RetrievalCapabilityError, + RetrievalSmokeError, + run_retrieval_smoke, +) +from powercontext_eval.benchmarks.longmemeval_v2.smoke import PreparedSmokeRun + + +class FakeRetrievalRuntime: + def __init__(self, *, fail_scope: str | None = None) -> None: + self.fail_scope = fail_scope + self.scopes: list[dict[str, object]] = [] + self.captures: list[dict[str, object]] = [] + self.memories: list[dict[str, object]] = [] + self.searches: list[dict[str, object]] = [] + self.text_by_scope: dict[str, list[str]] = {} + + def get_readiness(self) -> Mapping[str, object]: + return {"status": "ready"} + + def get_capabilities(self) -> Mapping[str, object]: + return { + "search_modes": ["auto", "fts"], + "context_versions": ["powercontext.prepared-context.v1"], + "source_types": ["content"], + "artifact_families": ["memory"], + } + + def create_scope(self, payload: Mapping[str, object]) -> Mapping[str, object]: + self.scopes.append(dict(payload)) + return {"scope_id": f"scope-{len(self.scopes)}"} + + def capture_content_source(self, payload: Mapping[str, object]) -> Mapping[str, object]: + request = dict(payload) + self.captures.append(request) + return {"source": {"name": "content", "source_id": request["source_id"]}} + + def remember_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + request = dict(payload) + if request["scope_id"] == self.fail_scope: + raise RuntimeError("selected group failed") + self.memories.append(request) + scope_id = str(request["scope_id"]) + self.text_by_scope.setdefault(scope_id, []).append(str(request["text"])) + index = len(self.memories) + return { + "entry": { + "citation": { + "memory_ref": {"family": "memory", "artifact_id": "memory", "revision": index}, + "entry_id": f"entry-{index}", + "entry_version_id": f"entry-{index}-v1", + } + } + } + + def search_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + request = dict(payload) + self.searches.append(request) + scope_id = str(request["scope_id"]) + text = self.text_by_scope[scope_id][0] + return { + "mode": request.get("mode"), + "hits": [ + { + "text": text, + "citation": { + "memory_ref": {"family": "memory", "artifact_id": "memory", "revision": 1}, + "entry_id": f"hit-{scope_id}", + "entry_version_id": f"hit-{scope_id}-v1", + }, + } + ], + } + + def prepare_context(self, payload: Mapping[str, object]) -> Mapping[str, object]: + request = dict(payload) + scope_id = str(request["scope_id"]) + maximum = request["max_bytes"] + assert isinstance(maximum, int) and not isinstance(maximum, bool) + content = self.text_by_scope[scope_id][0][:maximum] + return { + "schema": "powercontext.prepared-context.v1", + "status": "ready", + "content": content, + "content_bytes": len(content.encode()), + } + + +def write_inputs(tmp_path: Path, *, shared_haystack: bool = False) -> tuple[Path, Path]: + data_root = tmp_path / "data" + (data_root / "haystacks").mkdir(parents=True) + smoke_manifest = tmp_path / "smoke.json" + smoke_manifest.write_text( + json.dumps( + { + "schema": "powercontext.longmemeval-v2-smoke.v1", + "tier": "small", + "cases": [ + {"question_id": "question-a", "ability": "static_state"}, + {"question_id": "question-b", "ability": "workflow_knowledge"}, + ], + } + ), + encoding="utf-8", + ) + with (data_root / "questions.jsonl").open("w", encoding="utf-8") as stream: + stream.write( + json.dumps( + { + "id": "question-a", + "domain": "enterprise", + "question": "Where is enterprise evidence?", + "question_type": "must-not-be-copied", + "answer": "GOLD-MUST-NOT-LEAK", + } + ) + + "\n" + ) + stream.write( + json.dumps( + { + "id": "question-b", + "domain": "web", + "question": "Where is web evidence?", + "question_type": "must-not-be-copied", + "answer": "GOLD-MUST-NOT-LEAK", + } + ) + + "\n" + ) + with (data_root / "trajectories.jsonl").open("w", encoding="utf-8") as stream: + for trajectory_id, domain, evidence in ( + ("trajectory-a", "enterprise", "enterprise-only-evidence"), + ("decoy", "web", "decoy-evidence"), + ("trajectory-b", "web", "web-only-evidence"), + ): + stream.write( + json.dumps( + { + "id": trajectory_id, + "domain": domain, + "environment": "test", + "goal": evidence, + "outcome": "success", + "start_url": "https://example.test", + "states": [ + { + "url": "https://example.test", + "action": evidence, + "thought": evidence, + "accessibility_tree": evidence, + "screenshot": "unused.png", + } + ], + "gold_answer": "GOLD-MUST-NOT-LEAK", + "judge": {"secret": "GOLD-MUST-NOT-LEAK"}, + } + ) + + "\n" + ) + second = ["trajectory-a"] if shared_haystack else ["trajectory-b"] + (data_root / "haystacks" / "lme_v2_small.json").write_text( + json.dumps({"question-a": ["trajectory-a"], "question-b": second}), + encoding="utf-8", + ) + return data_root, smoke_manifest + + +def fake_preflight(**kwargs: object) -> PreparedSmokeRun: + output_dir = kwargs["output_dir"] + assert isinstance(output_dir, Path) + output_dir.mkdir(parents=True, exist_ok=False) + manifest = output_dir / "manifest.json" + subset = output_dir / "subset.json" + manifest.write_text("{}\n", encoding="utf-8") + subset.write_text("{}\n", encoding="utf-8") + return PreparedSmokeRun(output_dir=output_dir, manifest_path=manifest, subset_path=subset) + + +def run( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + runtime: FakeRetrievalRuntime, + *, + shared: bool = False, + **overrides: object, +) -> Path: + data_root, smoke_manifest = write_inputs(tmp_path, shared_haystack=shared) + output_dir = tmp_path / "output" + monkeypatch.setattr(retrieval_smoke, "prepare_smoke_run", fake_preflight) + digests = { + "questions.jsonl": hashlib.sha256((data_root / "questions.jsonl").read_bytes()).hexdigest(), + "trajectories.jsonl": hashlib.sha256((data_root / "trajectories.jsonl").read_bytes()).hexdigest(), + "haystacks/lme_v2_small.json": hashlib.sha256( + (data_root / "haystacks" / "lme_v2_small.json").read_bytes() + ).hexdigest(), + } + monkeypatch.setattr(retrieval_smoke, "load_dataset_lock", lambda path: SimpleNamespace(file_digests=digests)) + arguments: dict[str, Any] = { + "data_root": data_root, + "dataset_lock": tmp_path / "dataset-lock.json", + "harness_root": tmp_path / "harness", + "smoke_manifest": smoke_manifest, + "output_dir": output_dir, + "run_id": "test-run", + "powercontext_revision": "powercontext-sha", + "integration_revision": "integration-sha", + "runtime": runtime, + } + arguments.update(overrides) + run_retrieval_smoke(**arguments) + return output_dir + + +def read_jsonl(path: Path) -> list[dict[str, object]]: + return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines()] + + +def test_streams_selected_trajectories_into_isolated_haystack_scopes( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + original_read_text = Path.read_text + + def guarded_read_text(path: Path, encoding: str | None = None, errors: str | None = None) -> str: + if path.name == "trajectories.jsonl": + raise AssertionError("trajectories.jsonl must be streamed") + return original_read_text(path, encoding=encoding, errors=errors) + + monkeypatch.setattr(Path, "read_text", guarded_read_text) + runtime = FakeRetrievalRuntime() + output_dir = run(tmp_path, monkeypatch, runtime) + + assert len(runtime.scopes) == 3 + assert runtime.scopes[1]["parent_scope_id"] == "scope-1" + assert runtime.scopes[2]["parent_scope_id"] == "scope-1" + assert {request["scope_id"] for request in runtime.captures} == {"scope-2", "scope-3"} + assert all("decoy-evidence" not in str(request["content"]) for request in runtime.captures) + results = read_jsonl(output_dir / "retrieval-results.jsonl") + assert [result["question_id"] for result in results] == ["question-a", "question-b"] + assert [result["scope_id"] for result in results] == ["scope-2", "scope-3"] + assert all(result["status"] == "succeeded" for result in results) + assert "enterprise-only-evidence" in str(results[0]["memory_context"]) + assert "web-only-evidence" in str(results[1]["memory_context"]) + summary = json.loads((output_dir / "summary.json").read_text(encoding="utf-8")) + assert summary["question_count"] == 2 + assert summary["haystack_count"] == 2 + assert summary["trajectory_count"] == 2 + assert summary["succeeded"] == 2 + assert summary["accuracy"] is None + assert (output_dir / "failures.jsonl").read_text(encoding="utf-8") == "" + retained = "".join(path.read_text(encoding="utf-8") for path in output_dir.iterdir() if path.is_file()) + assert "GOLD-MUST-NOT-LEAK" not in retained + assert "must-not-be-copied" not in retained + + +def test_reuses_one_child_scope_and_one_ingest_for_an_identical_haystack( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + runtime = FakeRetrievalRuntime() + output_dir = run(tmp_path, monkeypatch, runtime, shared=True) + + assert len(runtime.scopes) == 2 + assert len(runtime.captures) == 1 + assert len(runtime.searches) == 2 + results = read_jsonl(output_dir / "retrieval-results.jsonl") + assert results[0]["scope_id"] == results[1]["scope_id"] == "scope-2" + + +class IdempotentScopeRuntime(FakeRetrievalRuntime): + """A Server that returns one Scope per idempotency key, like the real persistence layer.""" + + def __init__(self) -> None: + super().__init__() + self.scope_by_key: dict[str, str] = {} + + def create_scope(self, payload: Mapping[str, object]) -> Mapping[str, object]: + self.scopes.append(dict(payload)) + key = str(payload["idempotency_key"]) + if key not in self.scope_by_key: + self.scope_by_key[key] = f"scope-{len(self.scope_by_key) + 1}" + return {"scope_id": self.scope_by_key[key]} + + +def test_repeated_runs_with_the_same_run_id_and_haystack_get_isolated_scopes( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + runtime = IdempotentScopeRuntime() + data_root, smoke_manifest = write_inputs(tmp_path) + monkeypatch.setattr(retrieval_smoke, "prepare_smoke_run", fake_preflight) + digests = { + "questions.jsonl": hashlib.sha256((data_root / "questions.jsonl").read_bytes()).hexdigest(), + "trajectories.jsonl": hashlib.sha256((data_root / "trajectories.jsonl").read_bytes()).hexdigest(), + "haystacks/lme_v2_small.json": hashlib.sha256( + (data_root / "haystacks" / "lme_v2_small.json").read_bytes() + ).hexdigest(), + } + monkeypatch.setattr(retrieval_smoke, "load_dataset_lock", lambda path: SimpleNamespace(file_digests=digests)) + manifests = [] + for arm, directory in ( + ("current-memory-fts-v1", "output-fts"), + ("write-time-l0-l1-v1", "output-l0-l1"), + ): + run_retrieval_smoke( + data_root=data_root, + dataset_lock=tmp_path / "dataset-lock.json", + harness_root=tmp_path / "harness", + smoke_manifest=smoke_manifest, + output_dir=tmp_path / directory, + run_id="same-run", + powercontext_revision="powercontext-sha", + integration_revision="integration-sha", + experiment_arm=arm, + runtime=runtime, + ) + manifests.append(json.loads((tmp_path / directory / "retrieval-manifest.json").read_text(encoding="utf-8"))) + + assert manifests[0]["execution_namespace"] != manifests[1]["execution_namespace"] + assert manifests[0]["root_scope_id"] != manifests[1]["root_scope_id"] + first = {entry["scope_id"] for entry in manifests[0]["haystacks"]} + second = {entry["scope_id"] for entry in manifests[1]["haystacks"]} + assert first and second + assert not first & second + assert len(runtime.scope_by_key) == 6 + + +def test_classifies_one_haystack_ingest_failure_without_stopping_the_other( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + runtime = FakeRetrievalRuntime(fail_scope="scope-2") + output_dir = run(tmp_path, monkeypatch, runtime) + + results = read_jsonl(output_dir / "retrieval-results.jsonl") + assert [result["status"] for result in results] == ["failed", "succeeded"] + [failure] = read_jsonl(output_dir / "failures.jsonl") + assert failure["phase"] == "ingest" + assert failure["category"] == "integration" + summary = json.loads((output_dir / "summary.json").read_text(encoding="utf-8")) + assert summary["succeeded"] == 1 + assert summary["failed"] == 1 + + +def test_refuses_to_overwrite_before_contacting_powercontext(tmp_path: Path) -> None: + output_dir = tmp_path / "existing" + output_dir.mkdir() + runtime = FakeRetrievalRuntime() + + with pytest.raises(RetrievalSmokeError, match="Refusing to overwrite"): + run_retrieval_smoke( + data_root=tmp_path, + dataset_lock=tmp_path / "lock.json", + harness_root=tmp_path, + smoke_manifest=tmp_path / "smoke.json", + output_dir=output_dir, + run_id="test", + powercontext_revision="pc", + integration_revision="adapter", + runtime=runtime, + ) + + assert runtime.scopes == [] + + +class HybridRetrievalRuntime(FakeRetrievalRuntime): + """A Server whose capabilities advertise hybrid search; responses echo the requested mode.""" + + def get_capabilities(self) -> Mapping[str, object]: + return { + "search_modes": ["auto", "fts", "vector", "hybrid"], + "source_types": ["content"], + "artifact_families": ["memory"], + } + + +class SilentFtsFallbackRuntime(HybridRetrievalRuntime): + """A Server that claims hybrid support but silently executes fts search.""" + + def search_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + response = dict(super().search_memory(payload)) + response["mode"] = "fts" + return response + + +class MisreportingHybridRuntime(FakeRetrievalRuntime): + """A Server that reports hybrid execution for an explicitly requested fts search.""" + + def search_memory(self, payload: Mapping[str, object]) -> Mapping[str, object]: + response = dict(super().search_memory(payload)) + response["mode"] = "hybrid" + return response + + +def test_default_arm_keeps_the_previous_fts_behaviour(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + runtime = FakeRetrievalRuntime() + output_dir = run(tmp_path, monkeypatch, runtime) + + assert [request["mode"] for request in runtime.searches] == ["fts", "fts"] + manifest = json.loads((output_dir / "retrieval-manifest.json").read_text(encoding="utf-8")) + assert manifest["experiment_arm"] == { + "id": "current-memory-fts-v1", + "retrieval_strategy": "memory-search", + "search_mode": "fts", + "memory_projection": "deterministic-compact-v1", + "query_projection": "question-text-v1", + "prepared_context_max_bytes": None, + "temporal_filter": None, + "task_lens": None, + } + results = read_jsonl(output_dir / "retrieval-results.jsonl") + assert all(result["search_mode"] == {"requested": "fts", "actual": "fts"} for result in results) + summary = json.loads((output_dir / "summary.json").read_text(encoding="utf-8")) + assert summary["experiment_arm"]["id"] == "current-memory-fts-v1" + + +def test_hybrid_arm_searches_with_hybrid_mode_and_records_provenance( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + runtime = HybridRetrievalRuntime() + output_dir = run(tmp_path, monkeypatch, runtime, experiment_arm="current-memory-hybrid-v1") + + assert [request["mode"] for request in runtime.searches] == ["hybrid", "hybrid"] + manifest = json.loads((output_dir / "retrieval-manifest.json").read_text(encoding="utf-8")) + assert manifest["experiment_arm"]["id"] == "current-memory-hybrid-v1" + assert manifest["runtime"]["search_mode"] == "hybrid" + results = read_jsonl(output_dir / "retrieval-results.jsonl") + assert all(result["status"] == "succeeded" for result in results) + assert all(result["search_mode"] == {"requested": "hybrid", "actual": "hybrid"} for result in results) + summary = json.loads((output_dir / "summary.json").read_text(encoding="utf-8")) + assert summary["experiment_arm"]["id"] == "current-memory-hybrid-v1" + assert summary["succeeded"] == 2 + + +def test_query_time_compact_arm_uses_prepared_context_without_memory_search( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + runtime = FakeRetrievalRuntime() + output_dir = run(tmp_path, monkeypatch, runtime, experiment_arm="query-time-compact-v1") + + assert runtime.searches == [] + manifest = json.loads((output_dir / "retrieval-manifest.json").read_text(encoding="utf-8")) + assert manifest["experiment_arm"]["id"] == "query-time-compact-v1" + assert manifest["runtime"]["query_strategy"] == "prepared-context" + assert manifest["runtime"]["prepared_context_max_bytes"] == 8_000 + results = read_jsonl(output_dir / "retrieval-results.jsonl") + assert all(result["status"] == "succeeded" for result in results) + assert all(result["search_mode"] == {"requested": None, "actual": None} for result in results) + contexts = [result["memory_context"] for result in results] + assert all(isinstance(context, list) and len(context) == 1 for context in contexts) + + +def test_query_time_compact_arm_requires_prepared_context_capability( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + class MissingPreparedContextRuntime(FakeRetrievalRuntime): + def get_capabilities(self) -> Mapping[str, object]: + capabilities = dict(super().get_capabilities()) + capabilities["context_versions"] = [] + return capabilities + + runtime = MissingPreparedContextRuntime() + with pytest.raises(RetrievalCapabilityError, match="does not support PreparedContext"): + run(tmp_path, monkeypatch, runtime, experiment_arm="query-time-compact-v1") + + assert runtime.captures == [] + assert runtime.memories == [] + + +def test_write_time_l0_l1_arm_ingests_two_entries_per_trajectory( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + runtime = FakeRetrievalRuntime() + output_dir = run(tmp_path, monkeypatch, runtime, experiment_arm="write-time-l0-l1-v1") + + assert len(runtime.memories) == 4 + assert sum(str(memory["text"]).startswith("LongMemEval-V2 deterministic L0") for memory in runtime.memories) == 2 + assert sum(str(memory["text"]).startswith("LongMemEval-V2 deterministic L1") for memory in runtime.memories) == 2 + manifest = json.loads((output_dir / "retrieval-manifest.json").read_text(encoding="utf-8")) + assert manifest["experiment_arm"]["memory_projection"] == "deterministic-l0-l1-v1" + summary = json.loads((output_dir / "summary.json").read_text(encoding="utf-8")) + assert summary["succeeded"] == 2 + + +def test_task_lensed_arm_uses_a_deterministic_question_only_query( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + runtime = FakeRetrievalRuntime() + output_dir = run(tmp_path, monkeypatch, runtime, experiment_arm="task-lensed-selection-v1") + + assert len(runtime.searches) == 2 + assert runtime.searches[0]["query"] == "enterprise evidence" + assert runtime.searches[1]["query"] == "web evidence" + manifest = json.loads((output_dir / "retrieval-manifest.json").read_text(encoding="utf-8")) + assert manifest["experiment_arm"]["task_lens"] == "question-keywords-v1" + + +def test_hybrid_arm_fails_before_ingestion_when_the_server_lacks_the_capability( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + runtime = FakeRetrievalRuntime() # capabilities advertise only auto and fts + + with pytest.raises(RetrievalCapabilityError, match="does not support hybrid Memory search"): + run(tmp_path, monkeypatch, runtime, experiment_arm="current-memory-hybrid-v1") + + assert runtime.scopes == [] + assert runtime.captures == [] + assert not (tmp_path / "output").exists() + + +def test_hybrid_arm_records_an_integrity_failure_when_the_server_executed_fts( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + runtime = SilentFtsFallbackRuntime() + output_dir = run(tmp_path, monkeypatch, runtime, experiment_arm="current-memory-hybrid-v1") + + results = read_jsonl(output_dir / "retrieval-results.jsonl") + assert all(result["status"] == "failed" for result in results) + assert all(result["search_mode"] == {"requested": "hybrid", "actual": "fts"} for result in results) + failures = read_jsonl(output_dir / "failures.jsonl") + assert [failure["category"] for failure in failures] == ["integrity", "integrity"] + assert all(failure["error_type"] == "PowerContextMemoryModeError" for failure in failures) + summary = json.loads((output_dir / "summary.json").read_text(encoding="utf-8")) + assert summary["succeeded"] == 0 + assert summary["failed"] == 2 + + +def test_fts_arm_records_an_integrity_failure_when_the_server_executed_hybrid( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + runtime = MisreportingHybridRuntime() + output_dir = run(tmp_path, monkeypatch, runtime, experiment_arm="current-memory-fts-v1") + + results = read_jsonl(output_dir / "retrieval-results.jsonl") + assert all(result["status"] == "failed" for result in results) + assert all(result["search_mode"] == {"requested": "fts", "actual": "hybrid"} for result in results) + failures = read_jsonl(output_dir / "failures.jsonl") + assert [failure["category"] for failure in failures] == ["integrity", "integrity"] + summary = json.loads((output_dir / "summary.json").read_text(encoding="utf-8")) + assert summary["succeeded"] == 0 + assert summary["failed"] == 2 + + +def test_rejects_an_unregistered_experiment_arm(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + runtime = FakeRetrievalRuntime() + + with pytest.raises(ExperimentArmError, match="unknown experiment arm"): + run(tmp_path, monkeypatch, runtime, experiment_arm="l0-persistent-v1") + + assert runtime.scopes == [] diff --git a/evaluation/tests/unit/test_longmemeval_v2_run_smoke.py b/evaluation/tests/unit/test_longmemeval_v2_run_smoke.py new file mode 100644 index 0000000000..069898c314 --- /dev/null +++ b/evaluation/tests/unit/test_longmemeval_v2_run_smoke.py @@ -0,0 +1,575 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import json +from collections.abc import Callable, Mapping +from pathlib import Path +from typing import Any + +import pytest + +from powercontext_eval.benchmarks.longmemeval_v2.adapter import ( + PowerContextMemoryAdapterError, + PowerContextMemoryModeError, +) +from powercontext_eval.benchmarks.longmemeval_v2.arms import CURRENT_MEMORY_HYBRID, ExperimentArmError +from powercontext_eval.benchmarks.longmemeval_v2.costs import parse_cost_policy +from powercontext_eval.benchmarks.longmemeval_v2.prepare_smoke import PreparedPromptRun, PrepareSmokeError +from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import ReaderSmokeError, ReaderSmokeRun +from powercontext_eval.benchmarks.longmemeval_v2.replay_score import ReplayScoreRun +from powercontext_eval.benchmarks.longmemeval_v2.retrieval_smoke import ( + RetrievalCapabilityError, + RetrievalSmokeError, + RetrievalSmokeRun, +) +from powercontext_eval.benchmarks.longmemeval_v2.run_smoke import RunSmokeError, SmokeStages, classify_error, run_smoke +from powercontext_eval.benchmarks.longmemeval_v2.score_smoke import ScoreSmokeRun +from powercontext_eval.benchmarks.longmemeval_v2.smoke import PreparedSmokeRun + +SECRET = "sk-smoke-test-secret-value" +READER_PRICE_POLICY = { + "provider": "deepseek-openai", + "model": "deepseek-flash", + "currency": "USD", + "input_cache_hit_price_per_million": 0.006, + "input_cache_miss_price_per_million": 0.3, + "output_price_per_million": 1.2, + "price_policy_revision": "deepseek-public-list-2026-09", +} + + +class StubTransport: + """Stand in for a configured Reader or Judge transport without any network call.""" + + def complete(self, *, system: str, content: list[dict[str, object]]) -> Mapping[str, object]: + raise AssertionError("the orchestration tests must never call a model transport") + + +class _Harness: + """Record stage calls and emit the summary artifacts the run summary reads.""" + + def __init__(self, *, fail_on: str | None = None, correct: int = 4) -> None: + self.calls: list[str] = [] + self.retrieval_arguments: dict[str, Any] | None = None + self.reader_arguments: dict[str, Any] | None = None + self.score_arguments: dict[str, Any] | None = None + self._fail_on = fail_on + self._correct = correct + + def _record(self, name: str, arguments: dict[str, Any]) -> Path: + self.calls.append(name) + if name == "retrieval": + self.retrieval_arguments = dict(arguments) + if name == "reader": + self.reader_arguments = dict(arguments) + if name == "score": + self.score_arguments = dict(arguments) + if self._fail_on == name: + raise _stage_error(name) + output = Path(arguments["output_dir"]) + output.mkdir(parents=True, exist_ok=False) + return output + + def preflight(self, **arguments: Any) -> PreparedSmokeRun: + output = self._record("preflight", arguments) + manifest = output / "manifest.json" + manifest.write_text("{}\n", encoding="utf-8") + return PreparedSmokeRun(output_dir=output, manifest_path=manifest, subset_path=output / "subset.json") + + def retrieval(self, **arguments: Any) -> RetrievalSmokeRun: + output = self._record("retrieval", arguments) + _write( + output / "retrieval-results.jsonl", + [ + { + "question_id": f"q{index}", + "status": "succeeded", + "memory_context": [{"type": "text", "value": "a"}, {"type": "text", "value": "b"}], + "timings_ms": {"search": 1.0, "format": 0.5, "total": 2.0}, + } + for index in range(10) + ], + ) + _write( + output / "adapter-audit.jsonl", + [{"operation": "ingest", "timings_ms": {"source_capture": 1.0, "memory_remember": 1.0, "total": 3.0}}] * 4, + ) + _write_json( + output / "summary.json", + { + "question_count": 10, + "succeeded": 10, + "failed": 0, + "context_bytes": 900, + "citation_count": 12, + "elapsed_ms": 100.0, + }, + ) + return RetrievalSmokeRun( + output_dir=output, + manifest_path=output / "retrieval-manifest.json", + results_path=output / "retrieval-results.jsonl", + failures_path=output / "failures.jsonl", + summary_path=output / "summary.json", + audit_path=output / "adapter-audit.jsonl", + ) + + def prepare(self, **arguments: Any) -> PreparedPromptRun: + output = self._record("prepare", arguments) + _write_json( + output / "prepare-summary.json", + { + "question_count": 10, + "succeeded": 10, + "failed": 0, + "memory_context_tokens": 1234, + "elapsed_ms": 7.0, + }, + ) + return PreparedPromptRun( + output_dir=output, + manifest_path=output / "prepare-manifest.json", + prompts_path=output / "prepared-prompts.jsonl", + failures_path=output / "prepare-failures.jsonl", + summary_path=output / "prepare-summary.json", + ) + + def reader(self, **arguments: Any) -> ReaderSmokeRun: + output = self._record("reader", arguments) + _write(output / "reader-outputs.jsonl", [{"question_id": "q0", "reader_latency_ms": 5.0}] * 10) + _write_json( + output / "reader-summary.json", + { + "question_count": 10, + "succeeded": 10, + "failed": 0, + "usage": {"input_tokens": 100, "output_tokens": 20}, + "elapsed_ms": 50.0, + }, + ) + return ReaderSmokeRun( + output_dir=output, + manifest_path=output / "reader-manifest.json", + outputs_path=output / "reader-outputs.jsonl", + failures_path=output / "reader-failures.jsonl", + summary_path=output / "reader-summary.json", + ) + + def score(self, **arguments: Any) -> ScoreSmokeRun: + output = self._record("score", arguments) + incorrect = 10 - self._correct + _write_json( + output / "score-summary.json", + { + "question_count": 10, + "correct": self._correct, + "incorrect": incorrect, + "failed": 0, + "accuracy": self._correct / 10, + "judge_usage": {"input_tokens": 30, "output_tokens": 10}, + "elapsed_ms": 40.0, + }, + ) + return ScoreSmokeRun( + output_dir=output, + manifest_path=output / "score-manifest.json", + inputs_path=output / "scoring-inputs.local.jsonl", + results_path=output / "per-question.jsonl", + judge_outputs_path=output / "judge-outputs.jsonl", + failures_path=output / "score-failures.jsonl", + summary_path=output / "score-summary.json", + ) + + def replay(self, **arguments: Any) -> ReplayScoreRun: + output = self._record("replay", arguments) + incorrect = 10 - self._correct + _write_json( + output / "replay-summary.json", + { + "question_count": 10, + "correct": self._correct, + "incorrect": incorrect, + "failed": 0, + "accuracy": self._correct / 10, + }, + ) + return ReplayScoreRun( + output_dir=output, + manifest_path=output / "replay-manifest.json", + results_path=output / "replay-per-question.jsonl", + failures_path=output / "replay-failures.jsonl", + summary_path=output / "replay-summary.json", + ) + + def stages(self) -> SmokeStages: + return SmokeStages( + preflight=self.preflight, + retrieval=self.retrieval, + prepare=self.prepare, + reader=self.reader, + score=self.score, + replay=self.replay, + ) + + +def _stage_error(name: str) -> Exception: + return { + "preflight": PrepareSmokeError("preflight rejected the fixed inputs"), + "retrieval": ReaderSmokeError("retrieval could not reach the runtime"), + "prepare": PrepareSmokeError("the pinned harness worker failed"), + "reader": ReaderSmokeError("Reader request returned HTTP 401"), + "score": ReaderSmokeError("scoring failed"), + "replay": ReaderSmokeError("replay failed"), + }[name] + + +def _write(path: Path, rows: list[dict[str, object]]) -> None: + with path.open("w", encoding="utf-8", newline="") as stream: + for row in rows: + stream.write(json.dumps(row) + "\n") + + +def _write_json(path: Path, value: object) -> None: + path.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n", encoding="utf-8") + + +def _run(tmp_path: Path, harness: _Harness, **overrides: Any) -> Any: + lock = tmp_path / "dataset-lock.json" + manifest = tmp_path / "smoke.json" + lock.write_text('{"schema": "lock"}\n', encoding="utf-8") + manifest.write_text('{"schema": "smoke"}\n', encoding="utf-8") + arguments: dict[str, Any] = { + "data_root": tmp_path / "data", + "dataset_lock": lock, + "smoke_manifest": manifest, + "harness_root": tmp_path / "harness", + "harness_python": tmp_path / "harness-python", + "processor_revision": "processor-sha", + "output_dir": tmp_path / "run", + "powercontext_revision": "pc-sha", + "integration_revision": "integration-sha", + "stages": harness.stages(), + "reader_transport": StubTransport(), + "judge_transport": StubTransport(), + } + arguments.update(overrides) + return run_smoke(**arguments) + + +def test_run_smoke_runs_every_phase_in_order_and_writes_a_report(tmp_path: Path) -> None: + harness = _Harness() + result = _run(tmp_path, harness) + + assert result.status == "completed" + assert harness.calls == ["preflight", "retrieval", "prepare", "reader", "score", "replay"] + assert result.completed_phases == ("preflight", "retrieval", "prepare", "reader", "score", "replay") + assert result.skipped_phases == () + assert result.failed_phase is None + for name in ("01-inputs", "02-retrieval", "03-prepare", "04-reader", "05-score", "06-replay"): + assert (tmp_path / "run" / name).is_dir() + + summary = json.loads(result.summary_path.read_text(encoding="utf-8")) + assert summary["classification"] == "smoke-subset" + assert summary["status"] == "completed" + assert summary["accuracy"] == {"correct": 4, "incorrect": 6, "failed": 0, "value": 0.4} + assert summary["question_count"] == 10 + assert summary["artifacts"]["retrieval"] == "02-retrieval" + assert (tmp_path / "run" / "report.json").is_file() + assert (tmp_path / "run" / "report.md").is_file() + + +def test_run_smoke_records_the_manifest_before_any_stage_runs(tmp_path: Path) -> None: + harness = _Harness() + _run(tmp_path, harness) + + manifest = json.loads((tmp_path / "run" / "run-manifest.json").read_text(encoding="utf-8")) + assert manifest["classification"] == "smoke-subset" + assert manifest["run_id"] == "run" + assert manifest["modes"] == {"reader": True, "score": True} + assert manifest["experiment_arm"] == { + "id": "current-memory-fts-v1", + "retrieval_strategy": "memory-search", + "search_mode": "fts", + "memory_projection": "deterministic-compact-v1", + "query_projection": "question-text-v1", + "prepared_context_max_bytes": None, + "temporal_filter": None, + "task_lens": None, + } + assert manifest["powercontext"]["search_mode"] == "fts" + assert manifest["reader"]["model"] == "deepseek-flash" + assert manifest["reader"]["token_env"] == "DEEPSEEK_API_KEY" + assert manifest["judge"]["token_env"] == "DEEPSEEK_API_KEY" + assert manifest["revisions"] == {"powercontext": "pc-sha", "integration": "integration-sha"} + assert manifest["phases"]["replay"] == "06-replay" + assert manifest["privacy"]["credentials"] == "resolved-from-environment-at-runtime-never-recorded" + assert manifest["privacy"]["reference_answers"] == "never-read-by-the-adapter-retrieval-prepare-or-reader-stages" + assert "score stage reads the locked local questions" in manifest["privacy"]["reference_answers_readers"] + assert manifest["privacy"]["reference_answers_artifact"] == "05-score/scoring-inputs.local.jsonl" + assert harness.retrieval_arguments is not None + assert harness.retrieval_arguments["experiment_arm"].arm_id == "current-memory-fts-v1" + + +def test_run_smoke_refuses_an_existing_output_directory_before_any_stage(tmp_path: Path) -> None: + harness = _Harness() + (tmp_path / "run").mkdir() + + with pytest.raises(RunSmokeError, match="Refusing to overwrite smoke run artifacts"): + _run(tmp_path, harness) + + assert harness.calls == [] + + +def test_run_smoke_rejects_a_base_url_that_carries_credentials(tmp_path: Path) -> None: + harness = _Harness() + + with pytest.raises(RunSmokeError, match="must not contain credentials"): + _run(tmp_path, harness, powercontext_base_url="http://user:secret@127.0.0.1:8000") + + assert harness.calls == [] + assert not (tmp_path / "run").exists() + + +def test_run_smoke_stops_after_a_failed_phase_and_keeps_prior_artifacts(tmp_path: Path) -> None: + harness = _Harness(fail_on="prepare") + result = _run(tmp_path, harness) + + assert result.status == "failed" + assert result.failed_phase == "prepare" + assert harness.calls == ["preflight", "retrieval", "prepare"] + assert result.completed_phases == ("preflight", "retrieval") + assert (tmp_path / "run" / "01-inputs").is_dir() + assert (tmp_path / "run" / "02-retrieval").is_dir() + assert not (tmp_path / "run" / "04-reader").exists() + + failures = [json.loads(line) for line in result.failures_path.read_text(encoding="utf-8").splitlines()] + assert failures == [ + { + "schema": "powercontext.longmemeval-v2-smoke-run-failure.v1", + "phase": "prepare", + "error_class": "infrastructure", + "error_type": "PrepareSmokeError", + "summary": "the pinned harness worker failed", + } + ] + summary = json.loads(result.summary_path.read_text(encoding="utf-8")) + assert summary["status"] == "failed" + assert summary["accuracy"] is None + assert summary["failed_phase"] == "prepare" + + +def test_run_smoke_classifies_a_reader_failure_as_generation(tmp_path: Path) -> None: + harness = _Harness(fail_on="reader") + result = _run(tmp_path, harness) + + failures = [json.loads(line) for line in result.failures_path.read_text(encoding="utf-8").splitlines()] + assert failures[0]["error_class"] == "generation" + assert result.failed_phase == "reader" + assert result.completed_phases == ("preflight", "retrieval", "prepare") + + +def test_run_smoke_skips_model_phases_without_inventing_accuracy(tmp_path: Path) -> None: + harness = _Harness() + result = _run(tmp_path, harness, skip_reader=True) + + assert result.status == "partial" + assert harness.calls == ["preflight", "retrieval", "prepare"] + assert result.skipped_phases == ("reader", "score", "replay") + summary = json.loads(result.summary_path.read_text(encoding="utf-8")) + assert summary["accuracy"] is None + assert summary["completed_phases"] == ["preflight", "retrieval", "prepare"] + report = json.loads((tmp_path / "run" / "report.json").read_text(encoding="utf-8")) + assert report["classification"] == "smoke-subset" + assert report["accuracy"] is None + assert report["usage"]["reader_input_tokens"] is None + + +def test_run_smoke_requires_a_reader_token_before_any_model_work( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.delenv("DEEPSEEK_API_KEY", raising=False) + harness = _Harness() + result = _run(tmp_path, harness, reader_transport=None, judge_transport=None) + + assert result.status == "failed" + assert result.failed_phase == "preflight" + assert harness.calls == [] + assert (tmp_path / "run" / "run-manifest.json").is_file() + failures = [json.loads(line) for line in result.failures_path.read_text(encoding="utf-8").splitlines()] + assert failures[0]["error_class"] == "configuration" + assert "DEEPSEEK_API_KEY" in failures[0]["summary"] + + +def test_run_smoke_redacts_a_configured_secret_from_recorded_failures( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + monkeypatch.setenv("DEEPSEEK_API_KEY", SECRET) + harness = _Harness() + monkeypatch.setattr(harness, "reader", _leaky_reader(harness)) + result = _run(tmp_path, harness, reader_transport=None, judge_transport=None) + + assert result.failed_phase == "reader" + recorded = result.failures_path.read_text(encoding="utf-8") + assert SECRET not in recorded + assert "" in recorded + + +def test_run_smoke_redacts_a_short_configured_secret(tmp_path: Path, monkeypatch: pytest.MonkeyPatch) -> None: + short_secret = "abc" + monkeypatch.setenv("DEEPSEEK_API_KEY", short_secret) + harness = _Harness() + + def leaky_reader(**arguments: Any) -> Any: + raise ReaderSmokeError(f"Reader rejected bearer {short_secret}") + + monkeypatch.setattr(harness, "reader", leaky_reader) + result = _run(tmp_path, harness, reader_transport=None, judge_transport=None) + + assert result.failed_phase == "reader" + recorded = result.failures_path.read_text(encoding="utf-8") + assert short_secret not in recorded + assert "" in recorded + + +def test_run_smoke_redacts_the_powercontext_token_in_model_free_mode( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """The retrieval stage still resolves POWERCONTEXT_TOKEN when the Reader is skipped.""" + + monkeypatch.setenv("POWERCONTEXT_TOKEN", SECRET) + monkeypatch.delenv("DEEPSEEK_API_KEY", raising=False) + harness = _Harness() + monkeypatch.setattr(harness, "retrieval", _leaky_retrieval(harness)) + result = _run(tmp_path, harness, skip_reader=True) + + assert result.failed_phase == "retrieval" + recorded = result.failures_path.read_text(encoding="utf-8") + assert SECRET not in recorded + assert "" in recorded + + +def test_run_smoke_does_not_publish_completed_while_retrieval_is_still_pending( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """A skip-reader run must stay partial after preflight, not claim completion early.""" + + harness = _Harness() + observed: list[str] = [] + monkeypatch.setattr(harness, "retrieval", _observing_stage(harness.retrieval, observed, tmp_path / "run")) + result = _run(tmp_path, harness, skip_reader=True) + + assert observed == ["partial"] + assert result.status == "partial" + + +def test_run_smoke_does_not_publish_completed_while_model_phases_are_pending( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + """A full run stays partial until every phase, including replay, has completed.""" + + harness = _Harness() + observed: list[str] = [] + monkeypatch.setattr(harness, "reader", _observing_stage(harness.reader, observed, tmp_path / "run")) + result = _run(tmp_path, harness) + + assert observed == ["partial"] + assert result.status == "completed" + + +def _leaky_reader(harness: _Harness) -> Any: + def reader(**arguments: Any) -> Any: + harness._record("reader", arguments) + raise ReaderSmokeError(f"Reader rejected bearer {SECRET}") + + return reader + + +def _leaky_retrieval(harness: _Harness) -> Any: + def retrieval(**arguments: Any) -> Any: + harness._record("retrieval", arguments) + raise RetrievalSmokeError(f"Runtime rejected bearer {SECRET}") + + return retrieval + + +def _observing_stage(stage: Callable[..., Any], observed: list[str], run_dir: Path) -> Callable[..., Any]: + def wrapper(**arguments: Any) -> Any: + summary = json.loads((run_dir / "run-summary.json").read_text(encoding="utf-8")) + observed.append(summary["status"]) + return stage(**arguments) + + return wrapper + + +def test_run_smoke_forwards_the_selected_experiment_arm_to_retrieval(tmp_path: Path) -> None: + harness = _Harness() + result = _run(tmp_path, harness, skip_reader=True, experiment_arm="current-memory-hybrid-v1") + + assert result.status == "partial" + assert harness.retrieval_arguments is not None + assert harness.retrieval_arguments["experiment_arm"] is CURRENT_MEMORY_HYBRID + manifest = json.loads(result.manifest_path.read_text(encoding="utf-8")) + assert manifest["experiment_arm"]["id"] == "current-memory-hybrid-v1" + assert manifest["powercontext"]["search_mode"] == "hybrid" + report = json.loads((tmp_path / "run" / "report.json").read_text(encoding="utf-8")) + assert report["experiment_arm"]["id"] == "current-memory-hybrid-v1" + + +def test_run_smoke_rejects_an_unregistered_experiment_arm(tmp_path: Path) -> None: + harness = _Harness() + + with pytest.raises(ExperimentArmError, match="unknown experiment arm"): + _run(tmp_path, harness, experiment_arm="l0-persistent-v1") + + assert harness.calls == [] + assert not (tmp_path / "run").exists() + + +def test_run_smoke_classifies_mode_and_capability_errors(tmp_path: Path) -> None: + assert classify_error(PowerContextMemoryModeError("hybrid was not executed")) == "integrity" + assert classify_error(RetrievalCapabilityError("Server does not support hybrid")) == "infrastructure" + assert classify_error(RetrievalSmokeError("retrieval failed")) == "retrieval" + assert classify_error(PowerContextMemoryAdapterError("transport failed")) == "infrastructure" + + +def test_run_smoke_records_and_forwards_separate_reader_and_judge_price_policies(tmp_path: Path) -> None: + """Reader and Judge may run different models, so each stage carries its own policy.""" + + harness = _Harness() + reader_policy = parse_cost_policy(READER_PRICE_POLICY) + judge_policy = parse_cost_policy({**READER_PRICE_POLICY, "model": "deepseek-v4-pro"}) + assert reader_policy is not None and judge_policy is not None + result = _run(tmp_path, harness, reader_price_policy=reader_policy, judge_price_policy=judge_policy) + + manifest = json.loads(result.manifest_path.read_text(encoding="utf-8")) + assert manifest["cost_policy"] == { + "reader": READER_PRICE_POLICY, + "judge": {**READER_PRICE_POLICY, "model": "deepseek-v4-pro"}, + } + assert harness.reader_arguments is not None + assert harness.reader_arguments["price_policy"] is reader_policy + assert harness.score_arguments is not None + assert harness.score_arguments["judge_price_policy"] is judge_policy + + +def test_run_smoke_keeps_both_cost_policies_null_when_no_prices_are_configured(tmp_path: Path) -> None: + harness = _Harness() + result = _run(tmp_path, harness) + + manifest = json.loads(result.manifest_path.read_text(encoding="utf-8")) + assert manifest["cost_policy"] == {"reader": None, "judge": None} + assert harness.reader_arguments is not None + assert harness.reader_arguments["price_policy"] is None + assert harness.score_arguments is not None + assert harness.score_arguments["judge_price_policy"] is None diff --git a/evaluation/tests/unit/test_longmemeval_v2_score_smoke.py b/evaluation/tests/unit/test_longmemeval_v2_score_smoke.py new file mode 100644 index 0000000000..809a25ae57 --- /dev/null +++ b/evaluation/tests/unit/test_longmemeval_v2_score_smoke.py @@ -0,0 +1,308 @@ +# Copyright (c) 2026 OceanBase. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from __future__ import annotations + +import json +from collections.abc import Mapping +from pathlib import Path +from types import SimpleNamespace +from typing import Any + +import pytest + +from powercontext_eval.benchmarks.longmemeval_v2 import score_smoke +from powercontext_eval.benchmarks.longmemeval_v2.catalog import SmokeSelection +from powercontext_eval.benchmarks.longmemeval_v2.costs import ModelPricePolicy +from powercontext_eval.benchmarks.longmemeval_v2.reader_smoke import ReaderResponseError +from powercontext_eval.benchmarks.longmemeval_v2.score_smoke import ScoreSmokeError, run_score_smoke + + +class FakeMetrics: + def eval_name(self, spec: str) -> str: + return spec.split("|", 1)[0] + + def extract_boxed_answer(self, text: str) -> str: + return text + + def eval_from_spec(self, spec: str, prediction: str, answer: str) -> bool: + return prediction == answer and not spec.startswith("llm_") + + def score_to_bool(self, value: Any) -> bool: + return bool(value) + + def _build_abstention_judge_messages(self, **kwargs: Any) -> list[dict[str, str]]: + return [{"role": "system", "content": "abstention"}, {"role": "user", "content": kwargs["reference_answer"]}] + + def _build_gotchas_judge_messages(self, **kwargs: Any) -> list[dict[str, str]]: + return [{"role": "system", "content": "gotchas"}, {"role": "user", "content": kwargs["reference_answer"]}] + + def _parse_llm_binary_judgement(self, text: str) -> tuple[int, str]: + assert text == '{"label":1}' + return 1, "accepted" + + +class FakeJudge: + def __init__(self) -> None: + self.calls: list[tuple[str, list[dict[str, object]]]] = [] + + def complete(self, *, system: str, content: list[dict[str, object]]) -> Mapping[str, object]: + self.calls.append((system, content)) + return { + "content": [{"type": "text", "text": '{"label":1}'}], + "usage": {"input_tokens": 10, "output_tokens": 2}, + } + + +def score_fixture(tmp_path: Path) -> tuple[Path, Path, Path]: + reader = tmp_path / "reader" + data = tmp_path / "data" + reader.mkdir() + data.mkdir() + reader_manifest = {"classification": "smoke-subset-reader-only", "judge": None} + reader_summary = {"classification": "smoke-subset-reader-only", "question_count": 10, "failed": 0} + (reader / "reader-manifest.json").write_text(json.dumps(reader_manifest), encoding="utf-8") + (reader / "reader-summary.json").write_text(json.dumps(reader_summary), encoding="utf-8") + cases = [] + with ( + (reader / "reader-outputs.jsonl").open("w", encoding="utf-8") as outputs, + (data / "questions.jsonl").open("w", encoding="utf-8") as questions, + ): + for index in range(10): + question_id = f"q-{index}" + evaluator = "llm_abstention_checker" if index < 2 else "llm_gotchas_checker" if index < 4 else "exact" + answer = f"answer-{index}" + cases.append({"question_id": question_id, "ability": "static_state"}) + outputs.write(json.dumps({"question_id": question_id, "response_text": answer}) + "\n") + questions.write( + json.dumps( + { + "id": question_id, + "domain": "web", + "question": f"question {index}", + "answer": answer, + "eval_function": evaluator, + } + ) + + "\n" + ) + manifest = tmp_path / "smoke.json" + manifest.write_text( + json.dumps({"schema": "powercontext.longmemeval-v2-smoke.v1", "tier": "small", "cases": cases}), + encoding="utf-8", + ) + return reader, data, manifest + + +def allow_validated_catalog(monkeypatch: pytest.MonkeyPatch) -> None: + monkeypatch.setattr(score_smoke, "load_dataset_lock", lambda path: SimpleNamespace(tier="small", file_digests={})) + monkeypatch.setattr( + score_smoke, + "LongMemEvalV2Catalog", + SimpleNamespace( + load=lambda *args, **kwargs: SimpleNamespace( + select_smoke=lambda cases: SmokeSelection("small", tuple(cases)) + ) + ), + ) + + +def test_score_smoke_keeps_gold_only_in_local_scoring_inputs(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> None: + reader, data, manifest = score_fixture(tmp_path) + output = tmp_path / "score" + judge = FakeJudge() + monkeypatch.setattr(score_smoke, "validate_harness_checkout", lambda root: None) + monkeypatch.setattr(score_smoke, "_load_metrics", lambda root: FakeMetrics()) + allow_validated_catalog(monkeypatch) + + result = run_score_smoke( + reader_dir=reader, + data_root=data, + dataset_lock=tmp_path / "dataset-lock.json", + smoke_manifest=manifest, + harness_root=tmp_path / "harness", + output_dir=output, + judge_transport=judge, + ) + + summary = json.loads(result.summary_path.read_text(encoding="utf-8")) + assert summary["correct"] == 10 + assert summary["accuracy"] == 1.0 + assert summary["judge_calls"] == 4 + assert summary["judge_usage"] == {"input_tokens": 40, "output_tokens": 8} + assert len(judge.calls) == 4 + results = [json.loads(line) for line in result.results_path.read_text(encoding="utf-8").splitlines()] + assert {row["score_mode"] for row in results} == {"deterministic", "llm_judge"} + assert all("reference_answer" not in row for row in results) + scoring_inputs = result.inputs_path.read_text(encoding="utf-8") + assert "answer-0" in scoring_inputs + assert '"reference_answer"' not in result.results_path.read_text(encoding="utf-8") + + +def test_score_smoke_refuses_to_overwrite_before_reading_inputs(tmp_path: Path) -> None: + output = tmp_path / "score" + output.mkdir() + + with pytest.raises(ScoreSmokeError, match="Refusing to overwrite"): + run_score_smoke( + reader_dir=tmp_path / "missing", + data_root=tmp_path / "missing-data", + dataset_lock=tmp_path / "missing-lock", + smoke_manifest=tmp_path / "missing-smoke", + harness_root=tmp_path / "missing-harness", + output_dir=output, + ) + + +class UnparseableJudgementMetrics(FakeMetrics): + """The pinned parser rejects the Judge text; the completed call's usage must survive.""" + + def _parse_llm_binary_judgement(self, text: str) -> tuple[int, str]: + raise ValueError("judgement is malformed") + + +class CacheSplitJudge: + """A Judge stub whose responses carry usage with the cache split an explicit policy prices.""" + + def __init__(self) -> None: + self.calls: list[tuple[str, list[dict[str, object]]]] = [] + + def complete(self, *, system: str, content: list[dict[str, object]]) -> Mapping[str, object]: + self.calls.append((system, content)) + return { + "content": [{"type": "text", "text": "truncated judgement"}], + "usage": { + "input_tokens": 10, + "output_tokens": 2, + "input_cache_hit_tokens": 4, + "input_cache_miss_tokens": 6, + }, + } + + +def test_score_smoke_preserves_judge_usage_when_the_judgement_cannot_be_parsed( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + reader, data, manifest = score_fixture(tmp_path) + output = tmp_path / "score" + judge = CacheSplitJudge() + monkeypatch.setattr(score_smoke, "validate_harness_checkout", lambda root: None) + monkeypatch.setattr(score_smoke, "_load_metrics", lambda root: UnparseableJudgementMetrics()) + allow_validated_catalog(monkeypatch) + policy = ModelPricePolicy( + provider="deepseek-openai", + model="test-judge", + currency="USD", + input_cache_hit_price_per_million=1.0, + input_cache_miss_price_per_million=2.0, + output_price_per_million=3.0, + price_policy_revision="test-prices", + ) + + with pytest.raises(ScoreSmokeError, match="Scoring failed for 4"): + run_score_smoke( + reader_dir=reader, + data_root=data, + dataset_lock=tmp_path / "dataset-lock.json", + smoke_manifest=manifest, + harness_root=tmp_path / "harness", + output_dir=output, + judge_model="test-judge", + judge_price_policy=policy, + judge_transport=judge, + ) + + assert len(judge.calls) == 4 + summary = json.loads((output / "score-summary.json").read_text(encoding="utf-8")) + assert summary["judge_calls"] == 4 + assert summary["judge_usage"] == { + "input_tokens": 40, + "output_tokens": 8, + "input_cache_hit_tokens": 16, + "input_cache_miss_tokens": 24, + } + assert summary["judge_cost"]["cost_usd"] == pytest.approx(88 / 1_000_000) + failures = [json.loads(line) for line in (output / "score-failures.jsonl").read_text(encoding="utf-8").splitlines()] + assert len(failures) == 4 + assert all(failure["error_type"] == "JudgeJudgementError" for failure in failures) + assert all( + failure["judge_usage"] + == {"input_tokens": 10, "output_tokens": 2, "input_cache_hit_tokens": 4, "input_cache_miss_tokens": 6} + for failure in failures + ) + assert all(isinstance(failure["judge_latency_ms"], (int, float)) for failure in failures) + assert not (output / "judge-outputs.jsonl").read_text(encoding="utf-8") + + +def test_score_smoke_preserves_judge_usage_when_the_transport_returns_no_text( + monkeypatch: pytest.MonkeyPatch, tmp_path: Path +) -> None: + reader, data, manifest = score_fixture(tmp_path) + output = tmp_path / "score" + monkeypatch.setattr(score_smoke, "validate_harness_checkout", lambda root: None) + monkeypatch.setattr(score_smoke, "_load_metrics", lambda root: FakeMetrics()) + allow_validated_catalog(monkeypatch) + policy = ModelPricePolicy( + provider="deepseek-openai", + model="test-judge", + currency="USD", + input_cache_hit_price_per_million=1.0, + input_cache_miss_price_per_million=2.0, + output_price_per_million=3.0, + price_policy_revision="test-prices", + ) + + class NoTextJudge: + def complete(self, *, system: str, content: list[dict[str, object]]) -> Mapping[str, object]: + raise ReaderResponseError( + "Reader response contains no text", + usage={ + "input_tokens": 10, + "output_tokens": 2, + "input_cache_hit_tokens": 4, + "input_cache_miss_tokens": 6, + }, + ) + + with pytest.raises(ScoreSmokeError, match="Scoring failed for 4"): + run_score_smoke( + reader_dir=reader, + data_root=data, + dataset_lock=tmp_path / "dataset-lock.json", + smoke_manifest=manifest, + harness_root=tmp_path / "harness", + output_dir=output, + judge_model="test-judge", + judge_price_policy=policy, + judge_transport=NoTextJudge(), + ) + + summary = json.loads((output / "score-summary.json").read_text(encoding="utf-8")) + assert summary["judge_calls"] == 4 + assert summary["judge_usage"] == { + "input_tokens": 40, + "output_tokens": 8, + "input_cache_hit_tokens": 16, + "input_cache_miss_tokens": 24, + } + assert summary["judge_cost"]["cost_usd"] == pytest.approx(88 / 1_000_000) + failures = [json.loads(line) for line in (output / "score-failures.jsonl").read_text(encoding="utf-8").splitlines()] + assert len(failures) == 4 + assert all(failure["error_type"] == "JudgeJudgementError" for failure in failures) + assert all( + failure["judge_usage"] + == {"input_tokens": 10, "output_tokens": 2, "input_cache_hit_tokens": 4, "input_cache_miss_tokens": 6} + for failure in failures + ) + assert all(isinstance(failure["judge_latency_ms"], (int, float)) for failure in failures)