From 7f9c3642f25945882a9d57c754c429c55f84a2ed Mon Sep 17 00:00:00 2001 From: TsukiKage Date: Wed, 2 Sep 2026 21:55:48 +0800 Subject: [PATCH 1/4] docs(rfc): explainable PreparedContext receipts Add bilingual RFCs for an opt-in, ephemeral PreparedContext Receipt on prepare, without changing default injection or host recall. --- ...0_explainable_prepared_context_receipts.md | 350 ++++++++++++++++++ ...0_explainable_prepared_context_receipts.md | 314 ++++++++++++++++ zensical.toml | 2 + 3 files changed, 666 insertions(+) create mode 100644 docs/en/rfcs/0000_explainable_prepared_context_receipts.md create mode 100644 docs/zh/rfcs/0000_explainable_prepared_context_receipts.md diff --git a/docs/en/rfcs/0000_explainable_prepared_context_receipts.md b/docs/en/rfcs/0000_explainable_prepared_context_receipts.md new file mode 100644 index 0000000000..49d34118df --- /dev/null +++ b/docs/en/rfcs/0000_explainable_prepared_context_receipts.md @@ -0,0 +1,350 @@ +- Proposal Name: `explainable_prepared_context_receipts` +- Start Date: 2026-09-02 +- RFC PR: [oceanbase/powercontext#0000](https://github.com/oceanbase/powercontext/pull/0000) +- Tracking Issue: [oceanbase/powercontext#1356](https://github.com/oceanbase/powercontext/issues/1356) +- Related RFCs: [RFC 0014](0014_memory_layer_design.md), [RFC 0028](0028_context_pack.md), + [RFC 0046](0046_observability_foundations.md), and [RFC 0080](0080_memory_search_reranking.md) + +# Summary + +This RFC adds an optional, bounded PreparedContext Receipt to request-time recall. The existing +`powercontext.prepared-context.v1` injection value stays `status`, `content`, and `content_bytes`. Callers that opt in +receive a companion Receipt that identifies what the Runtime selected, what it omitted, which retrieval path it used, +and which byte budget it consumed, without retaining query text or Memory/Experience bodies. + +The first policy is `powercontext.prepared-context-receipt.v1`. Receipts are ephemeral diagnostics attached to one +`prepare` response. They are not Artifacts, have no Revision, are not persisted by default, and are not a second +authority for facts. OpenTelemetry spans remain the timing and outcome signal. A Receipt is the exact-selection +contract for one injected byte string. + +# Motivation + +`POST /v1/context/prepare` already owns selection, citation, rendering, and the UTF-8 budget. The Runtime keeps exact +origins on `PreparedContextBuild` and records search/build stage attributes on traces, then discards both at the public +boundary. Integrations and operators therefore see only an opaque injection string. + +That gap blocks three product uses: + +- An operator cannot tell whether an Agent received a decision, hit the budget, or received an empty result for a + different reason. +- Evaluation cannot score selected exact references, omission classes, or retrieval fallback independently from the + injected text. +- Adjacent work such as multi-resolution packing can compare policies only after a stable, content-free selection + record exists. + +Spans answer "which stage ran and how long it took". They do not identify the exact Memory entry versions or Experience +revisions that were rendered, and they must not grow into a public selection schema. Memory search's in-process rerank +trace is also the wrong surface: it describes one search, not the interleaved, budgeted PreparedContext that the host +injects. + +Without a Receipt, each host or benchmark will invent its own explanation of recall. That explanation will either leak +content or disagree with what was actually injected. + +# Guide-level explanation + +## Ask for a Receipt + +The default `prepare` request is unchanged: + +```python +prepared = await client.prepare_context( + PrepareContextRequest(scope_id="project:payments", query="Why did we choose SQLite?", max_bytes=8000) +) +``` + +The response remains `powercontext.prepared-context.v1`. Hosts that inject `content` keep their current parsing. + +A caller that needs to explain the same result sets `include_receipt` to true: + +```python +prepared = await client.prepare_context( + PrepareContextRequest( + scope_id="project:payments", + query="Why did we choose SQLite?", + max_bytes=8000, + include_receipt=True, + ) +) +``` + +When the Runtime produces a ready context, the response still contains the injection string and adds a Receipt. The +Receipt names the exact selected citations, groups omitted candidates by a closed reason enum, records the retrieval +mode actually used, and hashes the injected bytes. It never repeats the query or the selected bodies. + +Empty results also receive a Receipt when requested. An empty Receipt still reports the query digest, budget, retrieval +path, and omission counts so "no Memory" is distinguishable from "everything was over budget". + +Automatic host recall must not set `include_receipt`. Official Pi, DSH, OpenCode, Codex, Claude Code, and WorkBuddy +validators currently require the injection object to contain exactly `schema`, `status`, `content`, and `content_bytes`. +A default or host-recall Receipt would be rejected as an invalid response and would fail open without injecting +context. Operators, evaluation harnesses, and a later CLI request Receipts on a separate prepare call, or through an +updated validator that opts into the extra field. + +## Treat a Receipt as untrusted metadata + +A Receipt proves that PowerContext rendered those exact references under that policy and budget. It does not prove that +the historical content is currently true, and it does not outrank system, developer, repository, or current-user +instructions. Hosts must not inject Receipt JSON into the model prompt. The injection value remains `content`. + +PreparedContext Receipts are not Handoff Receipts. A Handoff Receipt is a Work Continuity acknowledgement of an exact +Handoff Revision. A PreparedContext Receipt explains one ephemeral recall. + +## Inspect without a new content API + +The compact Receipt is the first disclosure level. Progressive inspection reuses existing exact-read operations: + +1. Receipt: selected refs, omission counts, retrieval path, digest, budget. +2. Exact Memory or Experience read by the cited identity. +3. Exact Source evidence already attached to that Artifact, when the caller is authorized to read it. + +The Receipt does not cache item bodies for later expansion. If the caller needs the text, it loads the current exact +identity through the ordinary Memory and Experience APIs. If that identity has since been retired, the exact-read +failure is the explanation; the Receipt is not a time-travel store. + +## Failure stays fail-open + +Receipt assembly must not change or block injection. If `include_receipt` is true and Receipt construction fails, the +Server still returns the PreparedContext that would have been returned without the flag, omits `receipt`, and records a +content-free diagnostic. Integrations that ignore the new field keep working. + +# Reference-level explanation + +## Public request + +`PrepareContextRequest` gains one optional field: + +| Field | Default | Contract | +| --- | ---: | --- | +| `include_receipt` | `false` | When true, the Runtime attempts to attach `powercontext.prepared-context-receipt.v1` to this response. | + +Omitted, null, and false are equivalent. Existing clients send neither the field nor a Receipt parser. v1 does not +offer a Server deployment default that turns Receipts on for every prepare. + +`query`, `scope_id`, and `max_bytes` keep their current bounds. `query_digest` hashes the same normalized query Memory +search already uses (`normalize_text`: NFC Unicode of the trimmed query, UTF-8). The Receipt stores only `sha256:`. + +## Public response + +`PreparedContext` keeps `schema`, `status`, `content`, and `content_bytes`. It gains: + +| Field | Presence | Contract | +| --- | --- | --- | +| `receipt` | omitted unless requested and successfully built | `PreparedContextReceipt` | + +When `include_receipt` is false, null, or omitted, the JSON object must not contain a `receipt` key. A `null` value is +not equivalent to omission. Official host validators and OpenAPI `additionalProperties: false` reject unknown keys, +including `"receipt": null`. + +The injection schema name remains `powercontext.prepared-context.v1`. Default prepare responses therefore stay a +four-field object. Callers that set `include_receipt` true must parse the optional `receipt` field; official generated +clients are regenerated in the implementation PR. Host recall plugins keep their exact four-field validators until they +explicitly opt in. + +## Receipt schema + +Policy ID: `powercontext.prepared-context-receipt.v1`. + +```text +PreparedContextReceipt + schema: powercontext.prepared-context-receipt.v1 + receipt_id: opaque UUID + policy_id: powercontext.prepared-context-receipt.v1 + query_digest: sha256 hex of normalized query + content_digest: sha256 hex of injected UTF-8 content, or null when status=empty + requested_max_bytes: integer + used_bytes: integer, equal to content_bytes + truncated: true when any selected item was size-truncated + retrieval: + memory_mode: auto | fts | vector | hybrid | none + rerank_policy_id: string | null + rerank_fallback: boolean + experience_configured: boolean + selected: [SelectedItem] # schema max 16; current Builder emits at most 8 + omitted: [OmittedGroup] # max 16 groups + stages: [StageTiming] # memory.search, experience.search, context.build +``` + +`receipt_id` correlates this Receipt with the HTTP `X-PowerContext-Request-ID` in logs. It is not an Artifact ID and +must not be used as a durable fetch key in v1. + +`SelectedItem`: + +| Field | Contract | +| --- | --- | +| `kind` | `memory` or `experience` | +| `memory_citation` or `artifact_ref` | Exact identity already admitted by the Builder | +| `rendered_bytes` | UTF-8 size of that item's rendered fragment, not the source body | +| `truncated` | Whether the Builder truncated that item to fit | + +Selected items are listed in injection order. The set of selected identities must equal `PreparedContextBuild.origins`. +A Receipt whose selected refs do not match the injected origins is invalid and must not be returned; the Server then +follows the Receipt-assembly failure path. + +The OpenAPI array cap is 16 so a later Builder change does not require a schema bump. The current Coding Agent Builder +admits at most eight Memory items and two Experience items, interleaves Memory first, and caps the injected list at +eight. A v1 Receipt lists exactly that injected list, never a superset. + +`OmittedGroup`: + +| Field | Contract | +| --- | --- | +| `reason` | Closed enum below | +| `count` | Distinct dropped identities, except empty-set flags which always use `1` | + +Closed `reason` values: + +| Reason | Meaning | +| --- | --- | +| `duplicate` | Same Memory entry version or Experience revision already admitted | +| `blank` | Empty identity or empty renderable text | +| `family_limit` | Exceeded the Builder's Memory or Experience admission cap | +| `entry_limit` | Exceeded the combined injection item cap | +| `over_budget` | Could not fit even at the minimum truncated size | +| `rerank_not_selected` | Present in the coarse Memory pool and dropped by listwise rerank before the Builder | +| `memory_not_retrieved` | No Memory head, or Memory search returned zero hits | +| `experience_not_retrieved` | Experience recall is not configured, or it returned zero hits | + +`memory_not_retrieved` and `experience_not_retrieved` are empty-set flags, not scans of the Artifact. Each appears at +most once, with `count` equal to `1`. They are omitted when that source produced a non-empty candidate list, even if +every candidate is later dropped for another reason. Other reasons count distinct candidate identities in the pool that +reached the Builder or were excluded immediately before admission. Omission counts never try to count the entire Memory +Artifact. + +`StageTiming` records milliseconds for `memory.search`, `experience.search`, and `context.build`. Missing stages are +omitted. Stage names match the existing Runtime span names so a Receipt can be correlated with a trace without copying +span payloads. + +Rerank: when Memory search produced a `MemoryRerankTrace`, the Receipt copies `policy_id` and `used_fallback` only. It +does not copy candidate hits, selected ranks, usage, or any Memory text from that trace. + +## Bounds + +| Limit | Value | +| ---: | ---: | +| `selected` | 16 items in the schema; current Builder output is at most 8 | +| `omitted` groups | 16 | +| `stages` | 8 | +| `receipt` JSON UTF-8 size | 8192 bytes | +| item bodies, query text, prompts, vectors, secrets, tokens, absolute paths | forbidden | + +If a valid Receipt would exceed 8192 bytes, the Server drops the Receipt rather than truncating selected identities. +A truncated identity list would disagree with the injected bytes. + +## Persistence and MCP + +v1 does not persist Receipts and does not add a fetch-by-`receipt_id` operation. Evaluation that needs a durable record +stores the response itself in the evaluation harness. A later opt-in store with TTL requires its own RFC. + +`prepare_context` remains absent from the default MCP tool surface. Receipts are not a reason to project prepare as an +Agent-facing tool. + +## CLI + +The Client SDK exposes `include_receipt` on the existing prepare operation. A later CLI command such as +`powercontext context prepare --include-receipt` may print the Receipt as JSON. That command is implementation work +after this RFC; it must not inject Receipts into Agent prompts. + +## Compatibility + +| Surface | Change | +| --- | --- | +| Default `prepare` | Same four-field JSON object; no `receipt` key | +| OpenAPI `PrepareContextRequest` | Optional `include_receipt` | +| OpenAPI `PreparedContext` | Optional `receipt`, present only when requested and successfully built | +| SQLite / OceanBase | No schema change in v1 | +| Host recall plugins | No required change while they omit `include_receipt` | +| Host validators | Exact four-field checks remain valid on the default path | +| Tracing | No new required span; existing stage names are reused in `stages` | + +Generated Python, DSH, Pi, and OpenCode operation tables are regenerated in the implementation PR. Automatic recall +must keep sending the current request shape. A host that wants to log Receipts updates its validator in the same +change that sets `include_receipt`. + +## Implementation sketch + +1. Extend `PrepareContextRequest` and `PreparedContext` in OpenAPI, then regenerate bindings. +2. Keep `PreparedContextBuilder.build_result()` as the origin of selected items. Count omitted identities while + walking the same candidate lists the Builder already walks. +3. Copy retrieval mode and rerank summary from the Memory search result already produced in + `ScopedContextApplication._prepare`. +4. Hash the Memory-search-normalized query and the injected `content` after the Builder returns. +5. If Receipt validation fails, log a content-free error and return the four-field PreparedContext with no `receipt` + key. +6. Leave official host recall requests unchanged. Update a host validator only in a change that opts that host into + Receipts. + +Focused tests cover empty, ready, truncated, deduplicated, reranked, fallback, `include_receipt=false` (no `receipt` +key), `memory_not_retrieved` / `experience_not_retrieved` flags, and Receipt-assembly failure. Each ready case asserts +`content_digest` matches the returned `content` and selected refs match `origins`. Host recall fixtures keep asserting +exactly four response keys. + +# Drawbacks + +- Optional fields expand the OpenAPI models and every generated client even when most hosts never request a Receipt. +- Omission counts are summaries of the admitted candidate pool, not of the entire Memory Artifact, so they can be + misread as "how much of the project was considered". +- Callers might inject Receipt JSON into prompts despite the trust rule. +- `receipt_id` looks like a durable identifier even though v1 cannot fetch it later. + +# Rationale and alternatives + +**Opt-in field on `prepare`, not a second operation.** A separate `POST /v1/context/explain` would either re-run +selection and disagree with the injected bytes, or require the Server to remember the last prepare. The first is +incorrect; the second is persistence. Attaching the Receipt to the same response keeps one selection, one digest, and +no store. + +**Do not put diagnostics on PreparedContext v1 by default.** Every host would download selection metadata on the hot +path. Fail-open injectors would start depending on a larger schema. + +**Do not use OTel spans as the public contract.** Spans are sampled, exporter-specific, and must stay content-free. +They cannot carry exact Memory citations as a supported API. + +**Do not reuse Memory search's rerank trace.** That trace includes candidate hits and is in-process only. HTTP search +already withholds it. PreparedContext selection happens after Memory search and Experience search and applies a +different budget. + +**Do not persist every prepare.** Request-time recall would become an unbounded history of user queries. Evaluation can +keep the HTTP response in the harness. + +**Do not invent a progressive-content cache.** Expanding a Receipt item through existing exact-read APIs preserves +Artifact authority. A prepare-side body cache would be a second store of Memory text keyed by an ephemeral id. + +Impact of not doing this: hosts and benchmarks will keep reverse-engineering recall from injected text or private +logs, and later packing experiments will have no shared omission vocabulary. + +# Prior art + +PowerContext already has four related but distinct records: + +- `PreparedContextBuild.origins` is the exact selected set, discarded at the HTTP boundary. +- Runtime `memory.search`, `experience.search`, and `context.build` spans record counts and modes, not citations. +- `MemoryRerankTrace` explains listwise Memory search inside the process. +- Handoff Receipts acknowledge an exact Handoff Revision in Work Continuity. + +RFC 0028 already requires citations and budgets on the injection path; it does not expose the selection record. +RFC 0046 forbids putting Memory bodies and query text into traces. RFC 0080 keeps rerank diagnostics off the HTTP +search contract. + +Outside PowerContext, retrieval systems often return hit IDs and scores with the answer. This RFC returns exact +PowerContext identities and closed omission reasons, and it refuses scores and bodies so the diagnostic cannot become +another prompt. + +# Unresolved questions + +- Should a later evaluation profile persist Receipts under an explicit TTL, or is harness-side storage enough? +- Does Dashboard inspection belong in the first implementation issue, or only CLI/SDK? + +v1 keeps `include_receipt` request-only. A Server default that attaches Receipts to every prepare would break current +host validators and is out of scope. + +These remaining questions do not block accepting the v1 contract above. They belong in the implementation issue or a +follow-up RFC. + +# Future possibilities + +- A Context Inspector UI that renders selected refs and omission counts from a prepare-with-receipt response. +- Evaluation reports that join Receipt omission reasons with task scores. +- Multi-resolution packing ([#1426](https://github.com/oceanbase/powercontext/issues/1426)) reporting selected level + through the same Receipt `omitted` vocabulary. +- An opt-in durable Receipt store for audits, with query digests only and a short TTL. +- Host-visible, content-free diagnostics that log `receipt_id` and `content_digest` when recall is empty or truncated. diff --git a/docs/zh/rfcs/0000_explainable_prepared_context_receipts.md b/docs/zh/rfcs/0000_explainable_prepared_context_receipts.md new file mode 100644 index 0000000000..94ef306f9e --- /dev/null +++ b/docs/zh/rfcs/0000_explainable_prepared_context_receipts.md @@ -0,0 +1,314 @@ +- Proposal Name: `explainable_prepared_context_receipts` +- Start Date: 2026-09-02 +- RFC PR: [oceanbase/powercontext#0000](https://github.com/oceanbase/powercontext/pull/0000) +- Tracking Issue: [oceanbase/powercontext#1356](https://github.com/oceanbase/powercontext/issues/1356) +- Related RFCs: [RFC 0014](0014_memory_layer_design.md)、[RFC 0028](0028_context_pack.md)、 + [RFC 0046](0046_observability_foundations.md)、[RFC 0080](0080_memory_search_reranking.md) + +# Summary + +本 RFC 为请求时召回增加可选、有界的 PreparedContext Receipt。现有 `powercontext.prepared-context.v1` 注入值仍只有 +`status`、`content` 和 `content_bytes`。选择加入的调用方会得到一份伴随 Receipt,说明 Runtime 选中了什么、省略了什么、 +使用了哪条检索路径、消耗了多少 byte 预算,且不保留 query 原文或 Memory/Experience 正文。 + +首个策略为 `powercontext.prepared-context-receipt.v1`。Receipt 是附着在一次 `prepare` 响应上的短暂诊断,不是 +Artifact,没有 Revision,默认不持久化,也不是事实的第二权威。OpenTelemetry span 仍负责耗时与结果;Receipt 是一份 +注入字节串的精确选择契约。 + +# Motivation + +`POST /v1/context/prepare` 已经负责选择、引用、渲染和 UTF-8 预算。Runtime 在 `PreparedContextBuild` 上保留精确 +origins,并在 trace 上记录 search/build 阶段属性,但两者都在公开边界被丢弃。集成和运维因此只能看到一段不透明的 +注入字符串。 + +这个缺口会挡住三类产品用途: + +- 运维无法区分 Agent 是拿到了决策、撞上预算,还是因为别的原因得到空结果。 +- 评测无法在注入文本之外,单独给精确引用、省略类别或检索回退打分。 +- 多分辨率装箱等后续工作,只有在存在稳定、无正文的选择记录之后,才能比较策略。 + +Span 回答的是“哪一阶段跑了、花了多久”。它们不能作为公开 API 携带精确 Memory 引用,也不应膨胀成选择 schema。 +Memory 搜索的进程内 rerank trace 也不是正确表面:它描述一次搜索,而不是宿主实际注入的、经过交错和预算裁剪的 +PreparedContext。 + +没有 Receipt,每个宿主或 benchmark 都会自己发明召回解释。那种解释要么泄漏内容,要么和真正注入的字节对不上。 + +# Guide-level explanation + +## 请求一份 Receipt + +默认 `prepare` 请求不变: + +```python +prepared = await client.prepare_context( + PrepareContextRequest(scope_id="project:payments", query="Why did we choose SQLite?", max_bytes=8000) +) +``` + +响应仍是 `powercontext.prepared-context.v1`。注入 `content` 的宿主保持现有解析。 + +需要解释同一次结果的调用方将 `include_receipt` 设为 true: + +```python +prepared = await client.prepare_context( + PrepareContextRequest( + scope_id="project:payments", + query="Why did we choose SQLite?", + max_bytes=8000, + include_receipt=True, + ) +) +``` + +Runtime 生成 ready 上下文时,响应仍包含注入字符串,并附加 Receipt。Receipt 列出精确选中引用,按封闭枚举分组省略 +候选,记录实际使用的检索模式,并对注入字节做哈希。它不会重复 query 或被选中条目的正文。 + +请求 Receipt 时,空结果也会带 Receipt。空 Receipt 仍然报告 query digest、预算、检索路径和省略计数,从而区分 +“没有 Memory”和“全部超出预算”。 + +自动宿主召回不得设置 `include_receipt`。官方 Pi、DSH、OpenCode、Codex、Claude Code 和 WorkBuddy 校验器当前要求注入 +对象恰好包含 `schema`、`status`、`content` 和 `content_bytes`。默认或宿主召回带上 Receipt 会被当成非法响应,fail-open +且不注入上下文。运维、评测 harness 和后续 CLI 在另一次 prepare 上请求 Receipt,或使用已选择加入该字段的更新校验器。 + +## 把 Receipt 当作不可信元数据 + +Receipt 只证明 PowerContext 在该策略和预算下渲染了这些精确引用。它不证明历史内容当前为真,也不能压过 system、 +developer、仓库或当前用户指令。宿主不得把 Receipt JSON 注入模型 prompt。注入值仍然是 `content`。 + +PreparedContext Receipt 不是 Handoff Receipt。后者是 Work Continuity 对精确 Handoff Revision 的确认;前者解释一次 +短暂召回。 + +## 不增加新的内容 API 也能逐级查看 + +紧凑 Receipt 是第一层披露。更细的查看复用现有精确读取: + +1. Receipt:选中引用、省略计数、检索路径、digest、预算。 +2. 按引用身份精确读取 Memory 或 Experience。 +3. 在调用方有权读取时,查看该 Artifact 上已有的精确 Source 证据。 + +Receipt 不为以后展开而缓存条目正文。需要正文时,通过普通 Memory/Experience API 加载当前精确身份。若该身份已被 +retire,精确读取失败就是解释;Receipt 不是时光机。 + +## 失败保持 fail-open + +组装 Receipt 不得改变或阻断注入。`include_receipt` 为 true 但构造失败时,Server 仍返回未加该标志时本应返回的 +PreparedContext,省略 `receipt`,并记录无正文诊断。忽略新字段的集成继续可用。 + +# Reference-level explanation + +## 公开请求 + +`PrepareContextRequest` 增加一个可选字段: + +| 字段 | 默认 | 契约 | +| --- | ---: | --- | +| `include_receipt` | `false` | 为 true 时,Runtime 尝试在本次响应上附加 `powercontext.prepared-context-receipt.v1`。 | + +省略、null 和 false 等价。现有客户端既不发送该字段,也不解析 Receipt。v1 不提供把每次 prepare 都带上 Receipt 的 +Server 部署默认值。 + +`query`、`scope_id` 和 `max_bytes` 保持现有边界。`query_digest` 对 Memory 搜索已使用的同一规范化 query 做哈希 +(`normalize_text`:trimmed query 的 NFC Unicode,UTF-8)。Receipt 只保存 `sha256:`。 + +## 公开响应 + +`PreparedContext` 保留 `schema`、`status`、`content` 和 `content_bytes`,并增加: + +| 字段 | 出现条件 | 契约 | +| --- | --- | --- | +| `receipt` | 仅在请求且成功构造时出现 | `PreparedContextReceipt` | + +当 `include_receipt` 为 false、null 或省略时,JSON 对象不得包含 `receipt` 键。`null` 不等于省略。官方宿主校验器和 +OpenAPI `additionalProperties: false` 会拒绝未知键,包括 `"receipt": null`。 + +注入 schema 名称仍为 `powercontext.prepared-context.v1`。因此默认 prepare 响应仍是四字段对象。设置 +`include_receipt` 为 true 的调用方必须解析可选的 `receipt` 字段;官方生成客户端在实现 PR 中再生。宿主召回插件在 +明确选择加入之前,保持恰好四字段的校验。 + +## Receipt schema + +策略 ID:`powercontext.prepared-context-receipt.v1`。 + +```text +PreparedContextReceipt + schema: powercontext.prepared-context-receipt.v1 + receipt_id: opaque UUID + policy_id: powercontext.prepared-context-receipt.v1 + query_digest: 规范化 query 的 sha256 hex + content_digest: 注入 UTF-8 内容的 sha256 hex;status=empty 时为 null + requested_max_bytes: integer + used_bytes: integer,等于 content_bytes + truncated: 任一选中条目被按大小截断时为 true + retrieval: + memory_mode: auto | fts | vector | hybrid | none + rerank_policy_id: string | null + rerank_fallback: boolean + experience_configured: boolean + selected: [SelectedItem] # schema 最多 16;当前 Builder 最多产出 8 + omitted: [OmittedGroup] # 最多 16 组 + stages: [StageTiming] # memory.search, experience.search, context.build +``` + +`receipt_id` 用于在日志中与 HTTP `X-PowerContext-Request-ID` 关联。它不是 Artifact ID,v1 也不得把它当作可持久 +获取的键。 + +`SelectedItem`: + +| 字段 | 契约 | +| --- | --- | +| `kind` | `memory` 或 `experience` | +| `memory_citation` 或 `artifact_ref` | Builder 已接纳的精确身份 | +| `rendered_bytes` | 该条目渲染片段的 UTF-8 大小,不是源正文 | +| `truncated` | Builder 是否为适配预算截断了该条目 | + +选中条目按注入顺序排列。选中身份集合必须等于 `PreparedContextBuild.origins`。若 Receipt 的选中引用与注入 origins +不一致,则该 Receipt 无效,不得返回;Server 走 Receipt 组装失败路径。 + +OpenAPI 数组上限为 16,避免以后调整 Builder 时立刻改 schema。当前 Coding Agent Builder 最多接纳 8 条 Memory 和 +2 条 Experience,Memory 优先交错,注入列表上限为 8。v1 Receipt 列出的就是这份注入列表,不是超集。 + +`OmittedGroup`: + +| 字段 | 契约 | +| --- | --- | +| `reason` | 下列封闭枚举 | +| `count` | 被丢掉的候选身份数;空集标志固定为 `1` | + +封闭 `reason` 值: + +| 原因 | 含义 | +| --- | --- | +| `duplicate` | 同一 Memory entry version 或 Experience revision 已被接纳 | +| `blank` | 空身份或无可渲染文本 | +| `family_limit` | 超过 Builder 的 Memory 或 Experience 接纳上限 | +| `entry_limit` | 超过合计注入条目上限 | +| `over_budget` | 即使按最小截断大小也无法放入 | +| `rerank_not_selected` | 出现在粗排 Memory 池中,在进入 Builder 前被 listwise rerank 丢掉 | +| `memory_not_retrieved` | 没有 Memory head,或 Memory 搜索返回零命中 | +| `experience_not_retrieved` | 未配置 Experience 召回,或召回返回零命中 | + +`memory_not_retrieved` 和 `experience_not_retrieved` 是空集标志,不是对 Artifact 的扫描。各自最多出现一次,`count` +固定为 `1`。该来源已经产生非空候选列表时省略这组,即使这些候选后来因其他原因被丢掉。其他原因统计到达 Builder、 +或在接纳前立即排除的候选身份数。省略计数从不试图统计整个 Memory Artifact。 + +`StageTiming` 记录 `memory.search`、`experience.search` 和 `context.build` 的毫秒耗时。缺失阶段省略。阶段名与现有 +Runtime span 名一致,以便在不复制 span 载荷的情况下与 trace 关联。 + +Rerank:当 Memory 搜索产生 `MemoryRerankTrace` 时,Receipt 只复制 `policy_id` 和 `used_fallback`。不复制 candidate +hits、selected ranks、usage 或任何 Memory 文本。 + +## 边界 + +| 限制 | 取值 | +| ---: | ---: | +| `selected` | schema 16 条;当前 Builder 输出最多 8 条 | +| `omitted` 分组 | 16 | +| `stages` | 8 | +| `receipt` JSON UTF-8 大小 | 8192 bytes | +| 条目正文、query 原文、prompt、向量、密钥、token、绝对路径 | 禁止 | + +若一份合法 Receipt 将超过 8192 bytes,Server 丢弃 Receipt,而不是截断选中身份。截断后的身份列表会与注入字节不一致。 + +## 持久化与 MCP + +v1 不持久化 Receipt,也不新增按 `receipt_id` 获取的操作。需要耐久记录的评测在 harness 中保存响应本身。以后带 TTL +的可选存储需要单独 RFC。 + +`prepare_context` 仍不进入默认 MCP 工具面。Receipt 不是把 prepare 投影成 Agent 工具的理由。 + +## CLI + +Client SDK 在现有 prepare 操作上暴露 `include_receipt`。后续 CLI(例如 +`powercontext context prepare --include-receipt`)可以把 Receipt 打成 JSON。那是本 RFC 之后的实现工作;不得把 +Receipt 注入 Agent prompt。 + +## 兼容性 + +| 表面 | 变化 | +| --- | --- | +| 默认 `prepare` | 仍是四字段 JSON 对象;没有 `receipt` 键 | +| OpenAPI `PrepareContextRequest` | 可选 `include_receipt` | +| OpenAPI `PreparedContext` | 可选 `receipt`,仅在请求且成功构造时出现 | +| SQLite / OceanBase | v1 无 schema 变更 | +| 宿主召回插件 | 只要不发送 `include_receipt` 就无需修改 | +| 宿主校验器 | 默认路径上恰好四字段的检查仍然有效 | +| Tracing | 不要求新 span;现有阶段名复用到 `stages` | + +生成的 Python、DSH、Pi、OpenCode operation 表在实现 PR 中再生。自动召回必须保持当前请求形状。宿主若要记录 +Receipt,在同一改动里更新校验器并设置 `include_receipt`。 + +## 实现要点 + +1. 在 OpenAPI 中扩展 `PrepareContextRequest` 和 `PreparedContext`,再生成绑定。 +2. 继续以 `PreparedContextBuilder.build_result()` 作为选中条目来源。在 Builder 已遍历的同一候选列表上统计省略身份。 +3. 从 `ScopedContextApplication._prepare` 已得到的 Memory 搜索结果复制检索模式和 rerank 摘要。 +4. 在 Builder 返回后对 Memory 搜索已规范化的 query 和注入 `content` 做哈希。 +5. Receipt 校验失败时,记录无正文错误,并返回没有 `receipt` 键的四字段 PreparedContext。 +6. 官方宿主召回请求保持不变。只有在该宿主选择加入 Receipt 的同一改动里,才更新其校验器。 + +聚焦测试覆盖 empty、ready、truncated、deduplicated、reranked、fallback、`include_receipt=false`(没有 `receipt` +键)、`memory_not_retrieved` / `experience_not_retrieved` 标志,以及 Receipt 组装失败。每个 ready 用例断言 +`content_digest` 与返回的 `content` 一致,且选中引用与 `origins` 一致。宿主召回夹具继续断言响应恰好四个键。 + +# Drawbacks + +- 可选字段仍会扩大 OpenAPI 模型和所有生成客户端,即使多数宿主从不请求 Receipt。 +- 省略计数是已接纳候选池的摘要,不是整个 Memory Artifact 的摘要,可能被误读成“考虑了项目的多大一部分”。 +- 调用方仍可能不顾信任规则把 Receipt JSON 注入 prompt。 +- `receipt_id` 看起来像耐久标识,但 v1 无法稍后获取。 + +# Rationale and alternatives + +**在 `prepare` 上选择加入字段,而不是第二个操作。** 单独的 `POST /v1/context/explain` 要么重新跑选择并与注入字节 +不一致,要么要求 Server 记住上一次 prepare。前者不正确,后者就是持久化。把 Receipt 附在同一响应上,保证一次选择、 +一个 digest、无存储。 + +**不要默认把诊断放到 PreparedContext v1。** 每个宿主都会在热路径下载选择元数据。fail-open 注入器会开始依赖更大的 +schema。 + +**不要把 OTel span 当作公开契约。** Span 会被采样、依赖导出器,并且必须保持无正文。它们不能作为受支持 API 携带 +精确 Memory 引用。 + +**不要复用 Memory 搜索的 rerank trace。** 该 trace 包含 candidate hits,且仅在进程内。HTTP 搜索已经不返回它。 +PreparedContext 选择发生在 Memory 搜索和 Experience 搜索之后,并使用不同预算。 + +**不要持久化每一次 prepare。** 请求时召回会变成用户 query 的无界历史。评测可以在 harness 中保留 HTTP 响应。 + +**不要发明渐进内容缓存。** 通过现有精确读取 API 展开 Receipt 条目,能保持 Artifact 权威。prepare 侧按短暂 id 缓存 +正文会成为 Memory 文本的第二存储。 + +不做的影响:宿主和 benchmark 会继续从注入文本或私有日志反推召回,后续装箱实验也没有共享的省略词表。 + +# Prior art + +PowerContext 已有四种相关但不同的记录: + +- `PreparedContextBuild.origins` 是精确选中集合,在 HTTP 边界被丢弃。 +- Runtime 的 `memory.search`、`experience.search`、`context.build` span 记录计数和模式,不记录引用。 +- `MemoryRerankTrace` 解释进程内的 listwise Memory 搜索。 +- Handoff Receipt 在 Work Continuity 中确认精确 Handoff Revision。 + +RFC 0028 已经要求注入路径带引用和预算,但没有暴露选择记录。RFC 0046 禁止把 Memory 正文和 query 原文放进 +trace。RFC 0080 把 rerank 诊断留在 HTTP 搜索契约之外。 + +在 PowerContext 之外,检索系统常常随答案返回 hit ID 和分数。本 RFC 返回精确 PowerContext 身份和封闭省略原因, +并拒绝分数与正文,避免诊断变成另一段 prompt。 + +# Unresolved questions + +- 以后的评测配置是否应在显式 TTL 下持久化 Receipt,还是 harness 侧存储就够? +- Dashboard 查看属于第一个实现 Issue,还是只做 CLI/SDK? + +v1 保持 `include_receipt` 仅由请求开启。把 Receipt 附到每一次 prepare 的 Server 默认值会破坏当前宿主校验器,不在 +本 RFC 范围。 + +其余问题不阻止接受上面的 v1 契约。它们属于实现 Issue 或后续 RFC。 + +# Future possibilities + +- 从带 Receipt 的 prepare 响应渲染选中引用和省略计数的 Context Inspector UI。 +- 把 Receipt 省略原因与任务分数拼接的评测报告。 +- 多分辨率装箱([#1426](https://github.com/oceanbase/powercontext/issues/1426))通过同一套 Receipt `omitted` 词表 + 报告选中层级。 +- 面向审计的可选耐久 Receipt 存储,只保留 query digest 和短 TTL。 +- 召回为空或被截断时,宿主记录 `receipt_id` 和 `content_digest` 的无正文诊断。 diff --git a/zensical.toml b/zensical.toml index df523f0cf7..27b0267476 100644 --- a/zensical.toml +++ b/zensical.toml @@ -70,6 +70,7 @@ nav = [ { "RFCs" = [ { "Overview" = "en/rfcs/README.md" }, { "1345 Scope Organization and Agent Integration" = "en/rfcs/1345_scope_organization_and_agent_integration.md" }, + { "0000 Explainable PreparedContext Receipts" = "en/rfcs/0000_explainable_prepared_context_receipts.md" }, { "1229 Unified Workloads and Long-Horizon Memory Evaluation" = "en/rfcs/1229_unified_workloads_and_long_horizon_memory_evaluation.md" }, { "1223 Human-Agent Work Continuity" = "en/rfcs/1223_human_agent_work_continuity.md" }, { "0082 Handoff Report" = "en/rfcs/0082_handoff_report.md" }, @@ -158,6 +159,7 @@ nav = [ { "RFC" = [ { "概览" = "zh/rfcs/README.md" }, { "1345 Scope 组织与 Agent 集成" = "zh/rfcs/1345_scope_organization_and_agent_integration.md" }, + { "0000 可解释的 PreparedContext Receipt" = "zh/rfcs/0000_explainable_prepared_context_receipts.md" }, { "1229 统一工作负载与长程 Memory 评估" = "zh/rfcs/1229_unified_workloads_and_long_horizon_memory_evaluation.md" }, { "1223 人与 Agent 工作连续性" = "zh/rfcs/1223_human_agent_work_continuity.md" }, { "0082 Handoff 报告" = "zh/rfcs/0082_handoff_report.md" }, From 65120ceb3bc20a3c32e69125e1455c8f8c17d2aa Mon Sep 17 00:00:00 2001 From: TsukiKage Date: Wed, 2 Sep 2026 21:59:56 +0800 Subject: [PATCH 2/4] docs(rfc): number PreparedContext receipt RFC as 1435 Match the RFC filename and navigation entry to the pull request number. --- ...eipts.md => 1435_explainable_prepared_context_receipts.md} | 2 +- ...eipts.md => 1435_explainable_prepared_context_receipts.md} | 2 +- zensical.toml | 4 ++-- 3 files changed, 4 insertions(+), 4 deletions(-) rename docs/en/rfcs/{0000_explainable_prepared_context_receipts.md => 1435_explainable_prepared_context_receipts.md} (99%) rename docs/zh/rfcs/{0000_explainable_prepared_context_receipts.md => 1435_explainable_prepared_context_receipts.md} (99%) diff --git a/docs/en/rfcs/0000_explainable_prepared_context_receipts.md b/docs/en/rfcs/1435_explainable_prepared_context_receipts.md similarity index 99% rename from docs/en/rfcs/0000_explainable_prepared_context_receipts.md rename to docs/en/rfcs/1435_explainable_prepared_context_receipts.md index 49d34118df..cb1a9bea83 100644 --- a/docs/en/rfcs/0000_explainable_prepared_context_receipts.md +++ b/docs/en/rfcs/1435_explainable_prepared_context_receipts.md @@ -1,6 +1,6 @@ - Proposal Name: `explainable_prepared_context_receipts` - Start Date: 2026-09-02 -- RFC PR: [oceanbase/powercontext#0000](https://github.com/oceanbase/powercontext/pull/0000) +- RFC PR: [oceanbase/powercontext#1435](https://github.com/oceanbase/powercontext/pull/1435) - Tracking Issue: [oceanbase/powercontext#1356](https://github.com/oceanbase/powercontext/issues/1356) - Related RFCs: [RFC 0014](0014_memory_layer_design.md), [RFC 0028](0028_context_pack.md), [RFC 0046](0046_observability_foundations.md), and [RFC 0080](0080_memory_search_reranking.md) diff --git a/docs/zh/rfcs/0000_explainable_prepared_context_receipts.md b/docs/zh/rfcs/1435_explainable_prepared_context_receipts.md similarity index 99% rename from docs/zh/rfcs/0000_explainable_prepared_context_receipts.md rename to docs/zh/rfcs/1435_explainable_prepared_context_receipts.md index 94ef306f9e..8f54c7c8da 100644 --- a/docs/zh/rfcs/0000_explainable_prepared_context_receipts.md +++ b/docs/zh/rfcs/1435_explainable_prepared_context_receipts.md @@ -1,6 +1,6 @@ - Proposal Name: `explainable_prepared_context_receipts` - Start Date: 2026-09-02 -- RFC PR: [oceanbase/powercontext#0000](https://github.com/oceanbase/powercontext/pull/0000) +- RFC PR: [oceanbase/powercontext#1435](https://github.com/oceanbase/powercontext/pull/1435) - Tracking Issue: [oceanbase/powercontext#1356](https://github.com/oceanbase/powercontext/issues/1356) - Related RFCs: [RFC 0014](0014_memory_layer_design.md)、[RFC 0028](0028_context_pack.md)、 [RFC 0046](0046_observability_foundations.md)、[RFC 0080](0080_memory_search_reranking.md) diff --git a/zensical.toml b/zensical.toml index 27b0267476..fe0ab2d42a 100644 --- a/zensical.toml +++ b/zensical.toml @@ -70,7 +70,7 @@ nav = [ { "RFCs" = [ { "Overview" = "en/rfcs/README.md" }, { "1345 Scope Organization and Agent Integration" = "en/rfcs/1345_scope_organization_and_agent_integration.md" }, - { "0000 Explainable PreparedContext Receipts" = "en/rfcs/0000_explainable_prepared_context_receipts.md" }, + { "1435 Explainable PreparedContext Receipts" = "en/rfcs/1435_explainable_prepared_context_receipts.md" }, { "1229 Unified Workloads and Long-Horizon Memory Evaluation" = "en/rfcs/1229_unified_workloads_and_long_horizon_memory_evaluation.md" }, { "1223 Human-Agent Work Continuity" = "en/rfcs/1223_human_agent_work_continuity.md" }, { "0082 Handoff Report" = "en/rfcs/0082_handoff_report.md" }, @@ -159,7 +159,7 @@ nav = [ { "RFC" = [ { "概览" = "zh/rfcs/README.md" }, { "1345 Scope 组织与 Agent 集成" = "zh/rfcs/1345_scope_organization_and_agent_integration.md" }, - { "0000 可解释的 PreparedContext Receipt" = "zh/rfcs/0000_explainable_prepared_context_receipts.md" }, + { "1435 可解释的 PreparedContext Receipt" = "zh/rfcs/1435_explainable_prepared_context_receipts.md" }, { "1229 统一工作负载与长程 Memory 评估" = "zh/rfcs/1229_unified_workloads_and_long_horizon_memory_evaluation.md" }, { "1223 人与 Agent 工作连续性" = "zh/rfcs/1223_human_agent_work_continuity.md" }, { "0082 Handoff 报告" = "zh/rfcs/0082_handoff_report.md" }, From ca60d3db6e02e2f8eb8d17fa172b7db52504f198 Mon Sep 17 00:00:00 2001 From: TsukiKage Date: Mon, 5 Oct 2026 22:31:47 +0800 Subject: [PATCH 3/4] docs: complete prepared context receipt evidence contract --- ...5_explainable_prepared_context_receipts.md | 258 +++++++++++++----- ...5_explainable_prepared_context_receipts.md | 222 ++++++++++----- 2 files changed, 345 insertions(+), 135 deletions(-) diff --git a/docs/en/rfcs/1435_explainable_prepared_context_receipts.md b/docs/en/rfcs/1435_explainable_prepared_context_receipts.md index 0d56d7a122..7f8ac8dfa4 100644 --- a/docs/en/rfcs/1435_explainable_prepared_context_receipts.md +++ b/docs/en/rfcs/1435_explainable_prepared_context_receipts.md @@ -14,7 +14,7 @@ title: "RFC 1435: Explainable PreparedContext Receipts" This RFC adds an optional, bounded PreparedContext Receipt to request-time recall. The existing `powercontext.prepared-context.v1` injection value stays `status`, `content`, and `content_bytes`. Callers that opt in receive a companion Receipt that identifies what the Runtime selected, what it omitted, which retrieval path it used, -and which byte budget it consumed, without retaining query text or Memory/Experience bodies. +and which byte budget it consumed, without retaining query text or selected bodies. The first policy is `powercontext.prepared-context-receipt.v1`. Receipts are ephemeral diagnostics attached to one `prepare` response. They are not Artifacts, have no Revision, are not persisted by default, and are not a second @@ -98,11 +98,12 @@ Handoff Revision. A PreparedContext Receipt explains one ephemeral recall. The compact Receipt is the first disclosure level. Progressive inspection reuses existing exact-read operations: 1. Receipt: selected refs, omission counts, retrieval path, digest, budget. -2. Exact Memory or Experience read by the cited identity. +2. Exact Memory entry, Topic Memory, Experience, or Profile read by the Scope-qualified identity; code read by its + fingerprint, repository-relative path, file hash, and line range. 3. Exact Source evidence already attached to that Artifact, when the caller is authorized to read it. The Receipt does not cache item bodies for later expansion. If the caller needs the text, it loads the current exact -identity through the ordinary Memory and Experience APIs. If that identity has since been retired, the exact-read +identity through the ordinary Artifact and code APIs. If that identity has since been retired, the exact-read failure is the explanation; the Receipt is not a time-travel store. ## Failure stays fail-open @@ -153,86 +154,179 @@ PreparedContextReceipt schema: powercontext.prepared-context-receipt.v1 receipt_id: opaque UUID policy_id: powercontext.prepared-context-receipt.v1 - query_digest: sha256 hex of normalized query + query_digest: sha256 hex of normalized original query content_digest: sha256 hex of injected UTF-8 content, or null when status=empty requested_max_bytes: integer used_bytes: integer, equal to content_bytes truncated: true when any selected item was size-truncated + non_deterministic: true when model-backed rerank or query expansion was invoked retrieval: - memory_mode: auto | fts | vector | hybrid | none - rerank_policy_id: string | null - rerank_fallback: boolean - experience_configured: boolean - selected: [SelectedItem] # schema max 16; current Builder emits at most 8 - omitted: [OmittedGroup] # max 16 groups - stages: [StageTiming] # memory.search, experience.search, context.build + rounds: [RecallRound] # max 3: round 0 and at most 2 expansions + searches: [SearchGroup] # max 32 groups across all rounds + rerank_configs: [RerankConfig] # max 8 distinct configurations + reranks: [RerankGroup] # max 32 groups across all rounds + selected: [SelectedItem] # max 32, including code evidence + omitted: [OmittedGroup] # max 16 groups + stages: [StageTiming] # max 8 aggregate stage timings ``` `receipt_id` correlates this Receipt with the HTTP `X-PowerContext-Request-ID` in logs. It is not an Artifact ID and -must not be used as a durable fetch key in v1. +must not be used as a durable fetch key in v1. `non_deterministic` describes execution, not merely configuration: +model-backed rerank remains non-deterministic at temperature zero; a configured reranker that never ran does not set it. -`SelectedItem`: +### Exact selected identities -| Field | Contract | +| `SelectedItem.kind` | Required identity | | --- | --- | -| `kind` | `memory` or `experience` | -| `memory_citation` or `artifact_ref` | Exact identity already admitted by the Builder | -| `rendered_bytes` | UTF-8 size of that item's rendered fragment, not the source body | -| `truncated` | Whether the Builder truncated that item to fit | - -Selected items are listed in injection order. The set of selected identities must equal `PreparedContextBuild.origins`. -A Receipt whose selected refs do not match the injected origins is invalid and must not be returned; the Server then -follows the Receipt-assembly failure path. - -The OpenAPI array cap is 16 so a later Builder change does not require a schema bump. The current Coding Agent Builder -admits at most eight Memory items and two Experience items, interleaves Memory first, and caps the injected list at -eight. A v1 Receipt lists exactly that injected list, never a superset. - -`OmittedGroup`: - -| Field | Contract | -| --- | --- | -| `reason` | Closed enum below | -| `count` | Distinct dropped identities, except empty-set flags which always use `1` | +| `memory` | `memory_entry_address`: Scope-qualified Memory Artifact address, entry ID, entry version ID | +| `topic-memory` | `artifact_address`: Scope ID and exact Topic Memory Artifact ref | +| `experience` | `artifact_address`: Scope ID and exact Experience Artifact ref | +| `profile` | `artifact_address`: Scope ID and exact committed Profile Artifact ref | +| `code` | `code_evidence`: Scope ID, workspace fingerprint, repository-relative path, file SHA-256, inclusive start/end lines, snippet SHA-256 | + +An Artifact address contains `scope_id` and `artifact` (`family`, `artifact_id`, `revision`). Every selected item also +has `rendered_bytes` (UTF-8 size of its rendered fragment, excluding separators between items) and `truncated`. +The identity fields are mutually exclusive. Even a current-Scope Memory citation or Artifact ref is expanded to its +Scope-qualified address; the same Artifact or entry ID in two Scopes must remain two different identities. +`memory_entry_address` uses the existing `MemoryEntryAddress` shape (`memory`, `entry_id`, `entry_version_id`), with +`memory` an Artifact address. `code_evidence` uses the existing `CodeEvidenceRef` fields (`scope_id`, `fingerprint`, +`path`, `file_sha256`, `start_line`, `end_line`, `snippet_sha256`); it is not an Artifact ref. + +Selected items follow injection order. After normalizing short refs to addresses, their ordered identities must equal +`PreparedContextBuild.origins` followed by `PreparedContextBuild.code_origins`, including the final line range and +snippet hash of any truncated code item. A code-only ready result has empty `origins` and non-empty `code_origins`. +Neither a set comparison nor an `origins`-only comparison satisfies this invariant. Missing, extra, reordered, or +ambiguous identities invalidate the entire Receipt and follow the Receipt-assembly failure path. + +The 32-item cap covers the current explicit assembly maximum of 26 historical items (eight Memory, eight Topic Memory, +eight Profile, two Experience), plus up to four code items. The default builder's combined historical entry limit is +eight and it also supports Topic Memory; neither default is a universal selected-item bound. Runtime entry limits and +the byte budget still control actual selection. The Receipt never changes those limits to make its own schema fit. + +### Retrieval evidence across Scopes and rounds + +`RecallRound` contains `round` (0, 1, or 2), `outcome` (`completed` or `expansion_failed`), and `retained` (whether that +round's candidate pool contributes to the final build). It reuses the execution outcomes already tracked for recall +expansion. If expansion fails and the Runtime returns round-zero candidates, round zero is retained and every +abandoned expansion is marked not retained. An attempted but failed expansion is not reported as a successful search. +Expanded query text and its model response are never included. + +`SearchGroup` aggregates calls with the same `round`, `family`, actual `mode`, `outcome`, and `fallback_reason`, and +contains a positive `count`. Families are `memory`, `topic-memory`, `experience`, `profile`, and `code`; modes are +`fts`, `vector`, `hybrid`, `snapshot`, `code`, or `none`. `snapshot` describes a Profile read and `code` the existing code +query operation. `outcome` is `completed`, `not_run`, or `failed`; `none` means no retrieval mode was executed. An +Experience adapter must provide its actual retrieval mode; the Runtime must not infer it from the adapter's presence. +A disabled family has no group; a requested family with no configured reader or no searchable head has a `not_run` +group. A completed zero-hit search is still `completed` with its actual mode. `count` counts actual calls for completed or +failed groups, and skipped retrieval opportunities for not-run groups. Profile and code operations outside the +expansion loop belong to round zero and are counted once per actual invocation. + +`fallback_reason` is `none`, `inference_unavailable`, `inference_timeout`, or `reused_fts_fallback`. Choosing FTS normally +under `auto` uses `none`; dropping the vector channel after an inference error uses the observed error class. If a later +round reuses the FTS-only outcome of that failure, it uses `reused_fts_fallback`. Expansion failure belongs in +`RecallRound`, not in this search fallback enum. `auto` is a request policy, never an actual mode in a Receipt. + +For example, Memory searches in two authorized Scopes can produce these groups in the same round: + +```json +[ + {"round": 0, "family": "memory", "mode": "hybrid", "outcome": "completed", "fallback_reason": "none", "count": 1}, + {"round": 0, "family": "memory", "mode": "fts", "outcome": "completed", "fallback_reason": "inference_timeout", "count": 1} +] +``` -Closed `reason` values: +Grouping intentionally omits per-search Scope IDs: ContextReferences have no fixed count cap. Selected identities +always retain Scope, while aggregate counts describe all executed calls, including calls in rounds later abandoned. +Do not multiply Topic Memory, Profile, or code calls by the number of referenced Scopes: record the calls that actually +ran. Grouping is deterministic by the tuple of grouping fields, and cannot collapse different modes or fallback causes. + +### Rerank evidence + +`RerankConfig` has `config_id`, `policy_id`, `model`, `effective_settings`, `timeout_seconds`, `max_requests`, +`config_digest`, `prompt`, and `non_deterministic`. `policy_id` identifies the rerank instruction policy; +it is insufficient to identify the model or configuration. `model` is the credential-free provider/model identity (null for a declared deterministic, non-model reranker). +`effective_settings` contains the non-content model settings actually passed after inheritance, overrides, and +normalization, including explicit defaults such as temperature zero. It is not a dump of deployment configuration. +The implementation must define a versioned allowlist of safe setting names and types, and validate their values; +headers, credentials, URLs, arbitrary provider payloads, and content-bearing settings are excluded. + +`config_digest` is SHA-256 over RFC 8785 canonical JSON of a versioned record containing the policy, model, effective +non-content settings, timeout, request limit, and Prompt identity below. Safe execution-affecting settings cannot be +silently omitted from this record. An adapter whose effective configuration cannot be represented completely and safely +must omit the Receipt, with a content-free diagnostic, rather than claim exact evidence using a partial config digest. +This digest is configuration evidence, not a promise that the provider will reproduce identical results. + +`prompt` comes from the `ResolvedPrompt` bound to the actual invocation, not the latest Prompt head read afterward. +It contains `scope_id`, `key`, `definition_version`, `builtin_version`, `selection` (`built_in` or `artifact`), +`selected_version`, `compiled_digest`, and `artifact_address` (null for a built-in Prompt; Scope-qualified exact Prompt +Artifact address otherwise). Compiled instructions and demonstrations are excluded. Two Scopes with different Prompt +selections cannot share a config entry merely because the rerank instruction policy ID matches. + +`RerankGroup` contains `round`, `config_id`, `outcome` (`selected`, `fallback`, or `failed`), `fallback_reason` +(`none`, `empty_selection`, `inference_unavailable`, `inference_timeout`, or `invalid_output`), and positive `count`. +These fields summarize actual rerank invocations; identical configurations and outcomes are grouped and share one +config entry. A fallback still records the configuration that was invoked. No invocation means no group or config. +A model-backed config has `non_deterministic=true`, even when its invocation fails or falls back. An injected reranker +must supply equivalent evidence and declare whether it is model-backed; unavailable evidence is a Receipt failure, +not permission to invent a built-in Prompt or mark the call deterministic. A declared deterministic non-model +reranker uses `prompt=null` and `non_deterministic=false`, and identifies its actual algorithm/version and effective +safe settings in the configuration record. Merely lacking model metadata does not establish determinism. + +`config_id` is the `config_digest` itself; config entries are sorted by this digest and every rerank group must resolve +to exactly one entry. Group reranks by all four grouping fields and sort by that tuple. + +### Omission and timing summaries + +`OmittedGroup` contains `family`, `reason`, and positive `count`; equal family/reason pairs are grouped. Families use +the same five-value enum as selected items. Reasons are closed: | Reason | Meaning | | --- | --- | -| `duplicate` | Same Memory entry version or Experience revision already admitted | +| `duplicate` | Repeated Scope-qualified exact candidate identity excluded by deduplication | | `blank` | Empty identity or empty renderable text | -| `family_limit` | Exceeded the Builder's Memory or Experience admission cap | -| `entry_limit` | Exceeded the combined injection item cap | -| `over_budget` | Could not fit even at the minimum truncated size | -| `rerank_not_selected` | Present in the coarse Memory pool and dropped by listwise rerank before the Builder | -| `memory_not_retrieved` | No Memory head, or Memory search returned zero hits | -| `experience_not_retrieved` | Experience recall is not configured, or it returned zero hits | - -`memory_not_retrieved` and `experience_not_retrieved` are empty-set flags, not scans of the Artifact. Each appears at -most once, with `count` equal to `1`. They are omitted when that source produced a non-empty candidate list, even if -every candidate is later dropped for another reason. Other reasons count distinct candidate identities in the pool that -reached the Builder or were excluded immediately before admission. Omission counts never try to count the entire Memory -Artifact. - -`StageTiming` records milliseconds for `memory.search`, `experience.search`, and `context.build`. Missing stages are -omitted. Stage names match the existing Runtime span names so a Receipt can be correlated with a trace without copying -span payloads. - -Rerank: when Memory search produced a `MemoryRerankTrace`, the Receipt copies `policy_id` and `used_fallback` only. It -does not copy candidate hits, selected ranks, usage, or any Memory text from that trace. +| `family_limit` | Candidate excluded by that family's or assembly section's admission cap | +| `entry_limit` | Candidate excluded by the combined injection item cap | +| `below_min_bytes` | Candidate too short to truncate into the remaining budget | +| `no_fitting_truncation` | No permitted truncation fits the remaining byte budget | +| `rerank_not_selected` | Candidate in the coarse Memory pool excluded by listwise rerank before the Builder | +| `not_retrieved` | Requested family produced no candidates across the retained rounds | + +`below_min_bytes` and `no_fitting_truncation` preserve the Builder's existing `dropped_below_min_bytes` and +`dropped_no_fitting_truncation` distinction. Their sum is `dropped_items`; do not add that total again as a separate +omission. Successful truncation is recorded on the selected item and Receipt, not counted as an omitted candidate. + +`not_retrieved` is an empty-set flag per requested family with `count=1`, including a missing reader/head. It is absent +when that family produced any candidates for the final build, even if every candidate was later dropped. Other reasons +count exclusion events in the bounded candidate lists processed for the retained rounds and final build, not distinct +Artifacts in storage. A repeated identity increments `duplicate` when it is excluded. Abandoned expansion pools do not +inflate these counts. Reuse existing omission and recall-effort counters where they represent the same event; count +other exclusions while traversing the same bounded pools, without extra database searches or whole-Artifact scans. + +`StageTiming` contains `stage`, `duration_ms`, and `count`. Aggregate repeated instances of the same existing Runtime +stage by summing durations and recording the instance count; this sum need not equal wall-clock prepare latency. +Include only stages that ran, including abandoned rounds. Do not copy span payloads or add per-Scope timing lists. ## Bounds | Limit | Value | | ---: | ---: | -| `selected` | 16 items in the schema; current Builder output is at most 8 | -| `omitted` groups | 16 | -| `stages` | 8 | +| `selected` | 32 items | +| `retrieval.rounds` | 3 | +| `retrieval.searches` | 32 groups total | +| `retrieval.rerank_configs` | 8 distinct configurations | +| `retrieval.reranks` | 32 groups total | +| `omitted` | 16 groups | +| `stages` | 8 aggregate timings | | `receipt` JSON UTF-8 size | 8192 bytes | -| item bodies, query text, prompts, vectors, secrets, tokens, absolute paths | forbidden | +| item bodies, original or expanded query text, prompt bodies, model responses, vectors, secrets, tokens, absolute paths | forbidden | -If a valid Receipt would exceed 8192 bytes, the Server drops the Receipt rather than truncating selected identities. -A truncated identity list would disagree with the injected bytes. +Check the byte budget on the exact serialized Receipt returned to the caller. The count caps do not guarantee that +32 complete identities plus configuration evidence fit into 8192 bytes. Any count overflow, byte overflow, incomplete +identity, or missing execution evidence drops the entire Receipt with a content-free diagnostic and leaves injection +unchanged. Never truncate identities, replace configuration evidence with a generic policy ID, or ignore extra calls. +Implementation acceptance must measure representative mixed-family, cross-Scope, and rerank-config payloads using +real identity lengths. If ordinary workloads frequently exceed the byte budget, revise that budget using the measured +results before shipping; do not silently ship consistently absent diagnostics. ## Persistence and MCP @@ -266,21 +360,37 @@ change that sets `include_receipt`. ## Implementation sketch +Implementation begins only after this design is accepted, as required by the tracking issue. + 1. Extend `PrepareContextRequest` and `PreparedContext` in OpenAPI, then regenerate bindings. -2. Keep `PreparedContextBuilder.build_result()` as the origin of selected items. Count omitted identities while - walking the same candidate lists the Builder already walks. -3. Copy retrieval mode and rerank summary from the Memory search result already produced in - `ScopedContextApplication._prepare`. -4. Hash the Memory-search-normalized query and the injected `content` after the Builder returns. -5. If Receipt validation fails, log a content-free error and return the four-field PreparedContext with no `receipt` - key. +2. Preserve ordered selected identities and per-item rendering metadata from `build_scopes_result()` and code assembly. + Count exclusions in the same bounded pools and reuse existing budget-omission and recall-effort counters. +3. Carry request-local execution evidence out of each search and inference invocation before `_recall_scope` and + `_recall_round` discard it. Aggregate actual modes and fallback causes by round; retain failed expansion outcomes. + Bind rerank configuration evidence to the effective model settings and actual resolved Prompt at invocation time. +4. Hash the normalized original query and final injected `content` after the Builder returns. Validate ordered complete + origins, referential integrity of rerank groups/configs, all count caps, and exact serialized byte size. +5. On any Receipt failure, log a content-free diagnostic and return the unchanged PreparedContext without `receipt`. 6. Leave official host recall requests unchanged. Update a host validator only in a change that opts that host into - Receipts. - -Focused tests cover empty, ready, truncated, deduplicated, reranked, fallback, `include_receipt=false` (no `receipt` -key), `memory_not_retrieved` / `experience_not_retrieved` flags, and Receipt-assembly failure. Each ready case asserts -`content_digest` matches the returned `content` and selected refs match `origins`. Host recall fixtures keep asserting -exactly four response keys. + Receipts. Do not add a Receipt table or progressive-content cache. + +Implementation acceptance covers: + +- Empty, ready, truncated, deduplicated, reranked, fallback, and receipt-construction failure results; default, false, + and null flags continue to return exactly four response keys. +- Topic Memory in default recall; mixed Memory/Topic Memory/Experience/Profile assembly with 18 and 26 selected + historical items; code-only and mixed code results, including final truncated code ranges and hashes. +- Ordered complete selected identities across both origins collections, identical short IDs in different Scopes, + exact-read reuse, and `content_digest` equal to the hash of returned UTF-8 `content`. +- Hybrid and FTS in different Scopes in the same round, completed zero-hit versus not-run searches, normal FTS versus + inference fallback, reused FTS fallback, and failed expansion returning only the round-zero candidate pool. +- Effective inherited and separate rerank models/settings, distinct Scope Prompts and revisions, changed settings or + compiled Prompt changing the config digest, repeated config deduplication, temperature-zero non-determinism, and + injected rerankers with missing evidence causing fail-open Receipt omission. +- Both budget-drop sub-counts, all five families' empty-set flags, no double counting of abandoned rounds, each count + cap, representative serialized payload sizes, and byte overflow dropping the whole Receipt without identity loss. +- Equivalent Receipt semantics on SQLite and OceanBase, no schema migration, no additional recall/model calls caused + by `include_receipt`, and no forbidden content in successful Receipts or failure diagnostics. # Drawbacks @@ -320,7 +430,7 @@ logs, and later packing experiments will have no shared omission vocabulary. PowerContext already has four related but distinct records: -- `PreparedContextBuild.origins` is the exact selected set, discarded at the HTTP boundary. +- `PreparedContextBuild.origins` and `code_origins` retain the exact selected identities, discarded at the HTTP boundary. - Runtime `memory.search`, `experience.search`, and `context.build` spans record counts and modes, not citations. - `MemoryRerankTrace` explains listwise Memory search inside the process. - Handoff Receipts acknowledge an exact Handoff Revision in Work Continuity. diff --git a/docs/zh/rfcs/1435_explainable_prepared_context_receipts.md b/docs/zh/rfcs/1435_explainable_prepared_context_receipts.md index 02fe42ed47..7b3b50315f 100644 --- a/docs/zh/rfcs/1435_explainable_prepared_context_receipts.md +++ b/docs/zh/rfcs/1435_explainable_prepared_context_receipts.md @@ -13,7 +13,7 @@ title: "RFC 1435:可解释的 PreparedContext Receipt" 本 RFC 为请求时召回增加可选、有界的 PreparedContext Receipt。现有 `powercontext.prepared-context.v1` 注入值仍只有 `status`、`content` 和 `content_bytes`。选择加入的调用方会得到一份伴随 Receipt,说明 Runtime 选中了什么、省略了什么、 -使用了哪条检索路径、消耗了多少 byte 预算,且不保留 query 原文或 Memory/Experience 正文。 +使用了哪条检索路径、消耗了多少 byte 预算,且不保留 query 原文或被选中内容的正文。 首个策略为 `powercontext.prepared-context-receipt.v1`。Receipt 是附着在一次 `prepare` 响应上的短暂诊断,不是 Artifact,没有 Revision,默认不持久化,也不是事实的第二权威。OpenTelemetry span 仍负责耗时与结果;Receipt 是一份 @@ -87,10 +87,11 @@ PreparedContext Receipt 不是 Handoff Receipt。后者是 Work Continuity 对 紧凑 Receipt 是第一层披露。更细的查看复用现有精确读取: 1. Receipt:选中引用、省略计数、检索路径、digest、预算。 -2. 按引用身份精确读取 Memory 或 Experience。 +2. 按带 Scope 的身份精确读取 Memory entry、Topic Memory、Experience 或 Profile;代码则按 fingerprint、 + 仓库相对路径、文件哈希和行范围读取。 3. 在调用方有权读取时,查看该 Artifact 上已有的精确 Source 证据。 -Receipt 不为以后展开而缓存条目正文。需要正文时,通过普通 Memory/Experience API 加载当前精确身份。若该身份已被 +Receipt 不为以后展开而缓存条目正文。需要正文时,通过普通 Artifact/code API 加载当前精确身份。若该身份已被 retire,精确读取失败就是解释;Receipt 不是时光机。 ## 失败保持 fail-open @@ -138,80 +139,161 @@ PreparedContextReceipt schema: powercontext.prepared-context-receipt.v1 receipt_id: opaque UUID policy_id: powercontext.prepared-context-receipt.v1 - query_digest: 规范化 query 的 sha256 hex + query_digest: 规范化后的原始 query 的 sha256 hex content_digest: 注入 UTF-8 内容的 sha256 hex;status=empty 时为 null requested_max_bytes: integer used_bytes: integer,等于 content_bytes truncated: 任一选中条目被按大小截断时为 true + non_deterministic: 调用了模型 rerank 或 query 扩展时为 true retrieval: - memory_mode: auto | fts | vector | hybrid | none - rerank_policy_id: string | null - rerank_fallback: boolean - experience_configured: boolean - selected: [SelectedItem] # schema 最多 16;当前 Builder 最多产出 8 - omitted: [OmittedGroup] # 最多 16 组 - stages: [StageTiming] # memory.search, experience.search, context.build + rounds: [RecallRound] # 最多 3 轮:round 0 加至多 2 轮扩展 + searches: [SearchGroup] # 所有轮次合计最多 32 组 + rerank_configs: [RerankConfig] # 最多 8 个不同配置 + reranks: [RerankGroup] # 所有轮次合计最多 32 组 + selected: [SelectedItem] # 最多 32 条,包括代码证据 + omitted: [OmittedGroup] # 最多 16 组 + stages: [StageTiming] # 最多 8 个阶段耗时聚合 ``` -`receipt_id` 用于在日志中与 HTTP `X-PowerContext-Request-ID` 关联。它不是 Artifact ID,v1 也不得把它当作可持久 -获取的键。 +`receipt_id` 用于在日志中与 HTTP `X-PowerContext-Request-ID` 关联。它不是 Artifact ID,v1 不得将它当作耐久获取键。 +`non_deterministic` 描述实际执行,而非仅仅配置:模型 rerank 即使 temperature 为零仍是非确定性的;配置了但未运行的 +reranker 不会将该值设为 true。 -`SelectedItem`: +### 精确选中身份 -| 字段 | 契约 | +| `SelectedItem.kind` | 必需身份 | | --- | --- | -| `kind` | `memory` 或 `experience` | -| `memory_citation` 或 `artifact_ref` | Builder 已接纳的精确身份 | -| `rendered_bytes` | 该条目渲染片段的 UTF-8 大小,不是源正文 | -| `truncated` | Builder 是否为适配预算截断了该条目 | +| `memory` | `memory_entry_address`:带 Scope 的 Memory Artifact 地址、entry ID、entry version ID | +| `topic-memory` | `artifact_address`:Scope ID 与精确 Topic Memory Artifact ref | +| `experience` | `artifact_address`:Scope ID 与精确 Experience Artifact ref | +| `profile` | `artifact_address`:Scope ID 与精确已提交 Profile Artifact ref | +| `code` | `code_evidence`:Scope ID、workspace fingerprint、仓库相对路径、文件 SHA-256、包含端点的起止行、片段 SHA-256 | + +Artifact 地址包含 `scope_id` 和 `artifact`(`family`、`artifact_id`、`revision`)。每个选中条目还包含 +`rendered_bytes`(渲染片段的 UTF-8 大小,不含条目之间的分隔符)和 `truncated`。身份字段互斥。即使是当前 Scope 的 +Memory citation 或 Artifact ref,也要扩展成带 Scope 的地址;两个 Scope 中相同的 Artifact 或 entry ID 必须是两个身份。 +`memory_entry_address` 使用现有 `MemoryEntryAddress` 形状(`memory`、`entry_id`、`entry_version_id`), +其中 `memory` 是 Artifact 地址。`code_evidence` 使用现有 `CodeEvidenceRef` 字段(`scope_id`、`fingerprint`、 +`path`、`file_sha256`、`start_line`、`end_line`、`snippet_sha256`);它不是 Artifact ref。 + +选中条目按注入顺序排列。将简短 ref 规范化为地址后,其有序身份必须等于 `PreparedContextBuild.origins` 接上 +`PreparedContextBuild.code_origins`,包括截断代码条目最终的行范围和片段哈希。只有代码的 ready 结果会有空 `origins` +和非空 `code_origins`。集合比较或只比较 `origins` 都不满足此不变量。缺少、多出、重排或歧义身份使整份 Receipt 无效, +并进入 Receipt 组装失败路径。 + +32 条上限覆盖当前显式 assembly 最多 26 条历史内容(Memory、Topic Memory、Profile 各八条,Experience 两条), +再加最多四条代码。默认 Builder 的历史条目合计上限是八,且也支持 Topic Memory;这些默认值不是所有请求的选中上限。 +Runtime 条目限制和 byte 预算仍控制实际选择。Receipt 不得为适配自身 schema 而改变这些限制。 + +### 跨 Scope、跨轮次的检索证据 + +`RecallRound` 包含 `round`(0、1、2)、`outcome`(`completed` 或 `expansion_failed`)和 `retained`(该轮候选池 +是否参与最终构建)。它复用召回扩展已追踪的执行结果。扩展失败且 Runtime 返回 round-zero 候选时,round zero 被保留, +每个被放弃的扩展轮次都标记为未保留。尝试过但失败的扩展不得报告为成功搜索。扩展 query 原文和模型响应均不进入 Receipt。 + +`SearchGroup` 聚合 `round`、`family`、实际 `mode`、`outcome`、`fallback_reason` 相同的调用,包含正整数 `count`。 +family 为 `memory`、`topic-memory`、`experience`、`profile`、`code`;mode 为 `fts`、`vector`、`hybrid`、 +`snapshot`、`code` 或 `none`。`snapshot` 描述 Profile 读取,`code` 描述现有代码 query 操作。 +`outcome` 为 `completed`、`not_run` 或 `failed`;`none` 表示没有执行检索模式。Experience adapter 必须提供实际模式, +Runtime 不得根据 adapter 是否存在来推断。未启用的 family 不产生组;请求了但没有配置 reader 或可搜索 head 的 family +产生 `not_run` 组。已完成但零命中的搜索仍为 `completed`,保留实际模式。 +completed/failed 组的 `count` 统计实际调用;not-run 组统计跳过的检索机会。扩展循环外的 Profile 和 code 操作归入 +round zero,每次实际调用只计一次。 + +`fallback_reason` 为 `none`、`inference_unavailable`、`inference_timeout` 或 `reused_fts_fallback`。 +`auto` 正常选择 FTS 使用 `none`;推理错误后放弃 vector 通道则使用实际错误类别。后续轮次复用该失败留下的 FTS-only +结果时使用 `reused_fts_fallback`。扩展失败属于 `RecallRound`,不属于搜索回退枚举。`auto` 是请求策略,不是 Receipt +中的实际执行模式。 + +例如,同一轮对两个已授权 Scope 的 Memory 搜索可以产生以下组: + +```json +[ + {"round": 0, "family": "memory", "mode": "hybrid", "outcome": "completed", "fallback_reason": "none", "count": 1}, + {"round": 0, "family": "memory", "mode": "fts", "outcome": "completed", "fallback_reason": "inference_timeout", "count": 1} +] +``` -选中条目按注入顺序排列。选中身份集合必须等于 `PreparedContextBuild.origins`。若 Receipt 的选中引用与注入 origins -不一致,则该 Receipt 无效,不得返回;Server 走 Receipt 组装失败路径。 +聚合有意不列每次搜索的 Scope ID:ContextReferences 没有固定数量上限。选中身份始终保留 Scope;聚合计数描述所有实际 +执行的调用,包括最终放弃的轮次。不得把 Topic Memory、Profile 或 code 的调用数乘以引用 Scope 数;只记录实际执行的 +调用。按分组字段元组确定性排序,且不得合并不同模式或回退原因。 -OpenAPI 数组上限为 16,避免以后调整 Builder 时立刻改 schema。当前 Coding Agent Builder 最多接纳 8 条 Memory 和 -2 条 Experience,Memory 优先交错,注入列表上限为 8。v1 Receipt 列出的就是这份注入列表,不是超集。 +### Rerank 证据 -`OmittedGroup`: +`RerankConfig` 包含 `config_id`、`policy_id`、`model`、`effective_settings`、`timeout_seconds`、`max_requests`、 +`config_digest`、`prompt` 和 `non_deterministic`。`policy_id` 标识 rerank 指令策略,不足以标识模型或配置。 +`model` 是不带凭据的 provider/model 身份(明确声明为确定性、非模型 reranker 时为 null)。`effective_settings` 包含继承、覆盖和规范化后实际传入的非正文模型参数, +包括 temperature 为零等显式默认值;不是部署配置的完整转储。实现必须定义带版本的安全参数名和类型白名单,并校验值; +header、凭据、URL、任意 provider payload 和承载正文的参数均排除。 -| 字段 | 契约 | -| --- | --- | -| `reason` | 下列封闭枚举 | -| `count` | 被丢掉的候选身份数;空集标志固定为 `1` | +`config_digest` 是带版本记录经 RFC 8785 canonical JSON 编码后的 SHA-256;该记录包含策略、模型、有效非正文参数、 +timeout、request limit 和下述 Prompt 身份。影响执行的安全参数不得被悄悄省略。adapter 若无法完整、安全地表达有效 +配置,必须省略 Receipt 并记录无正文诊断,不能用部分配置 digest 声称精确证据。该 digest 是配置证据,不保证 provider +重现相同结果。 -封闭 `reason` 值: +`prompt` 来自实际调用已绑定的 `ResolvedPrompt`,不能在调用结束后读取最新 Prompt head 代替。它包含 `scope_id`、 +`key`、`definition_version`、`builtin_version`、`selection`(`built_in` 或 `artifact`)、`selected_version`、 +`compiled_digest` 和 `artifact_address`(内置 Prompt 为 null;否则是带 Scope 的精确 Prompt Artifact 地址)。 +编译后的指令和 demonstrations 排除。两个 Scope 的 Prompt 选择不同,不能仅因 rerank 指令策略 ID 相同就共用配置条目。 -| 原因 | 含义 | -| --- | --- | -| `duplicate` | 同一 Memory entry version 或 Experience revision 已被接纳 | -| `blank` | 空身份或无可渲染文本 | -| `family_limit` | 超过 Builder 的 Memory 或 Experience 接纳上限 | -| `entry_limit` | 超过合计注入条目上限 | -| `over_budget` | 即使按最小截断大小也无法放入 | -| `rerank_not_selected` | 出现在粗排 Memory 池中,在进入 Builder 前被 listwise rerank 丢掉 | -| `memory_not_retrieved` | 没有 Memory head,或 Memory 搜索返回零命中 | -| `experience_not_retrieved` | 未配置 Experience 召回,或召回返回零命中 | +`RerankGroup` 包含 `round`、`config_id`、`outcome`(`selected`、`fallback` 或 `failed`)、`fallback_reason` +(`none`、`empty_selection`、`inference_unavailable`、`inference_timeout` 或 `invalid_output`)和正整数 `count`。 +这些字段汇总实际 rerank 调用;相同配置和结果进行聚合并共用一个配置条目。回退仍记录调用过的配置。没有调用就没有组或 +配置。模型配置的 `non_deterministic=true`,即使调用失败或回退也一样。注入的 reranker 必须提供等价证据,并声明是否 +使用模型;证据不可得就是 Receipt 失败,不能虚构内置 Prompt 或把调用标成确定性。明确声明的确定性非模型 +reranker 使用 `prompt=null`、`non_deterministic=false`,并在配置记录中标识实际算法/版本及有效安全参数。 +仅仅缺少模型元数据不代表确定性。 -`memory_not_retrieved` 和 `experience_not_retrieved` 是空集标志,不是对 Artifact 的扫描。各自最多出现一次,`count` -固定为 `1`。该来源已经产生非空候选列表时省略这组,即使这些候选后来因其他原因被丢掉。其他原因统计到达 Builder、 -或在接纳前立即排除的候选身份数。省略计数从不试图统计整个 Memory Artifact。 +`config_id` 就是 `config_digest`;配置条目按 digest 排序,每个 rerank 组必须恰好引用一个条目。rerank 按四个 +分组字段聚合,并按该元组排序。 -`StageTiming` 记录 `memory.search`、`experience.search` 和 `context.build` 的毫秒耗时。缺失阶段省略。阶段名与现有 -Runtime span 名一致,以便在不复制 span 载荷的情况下与 trace 关联。 +### 省略与耗时摘要 -Rerank:当 Memory 搜索产生 `MemoryRerankTrace` 时,Receipt 只复制 `policy_id` 和 `used_fallback`。不复制 candidate -hits、selected ranks、usage 或任何 Memory 文本。 +`OmittedGroup` 包含 `family`、`reason` 和正整数 `count`;相同 family/reason 聚合。family 与选中条目共用五值枚举。 +reason 为封闭枚举: + +| 原因 | 含义 | +| --- | --- | +| `duplicate` | 去重排除了重复的、带 Scope 的精确候选身份 | +| `blank` | 空身份或无可渲染文本 | +| `family_limit` | 候选超过该 family 或 assembly section 的接纳上限 | +| `entry_limit` | 候选超过合计注入条目上限 | +| `below_min_bytes` | 候选太短,无法截断进剩余预算 | +| `no_fitting_truncation` | 没有允许的截断方式能放入剩余 byte 预算 | +| `rerank_not_selected` | 粗排 Memory 池中的候选在进入 Builder 前被 listwise rerank 排除 | +| `not_retrieved` | 请求的 family 在保留轮次中没有产生候选 | + +`below_min_bytes` 和 `no_fitting_truncation` 保留 Builder 已有的 `dropped_below_min_bytes` 与 +`dropped_no_fitting_truncation` 区别。两者相加等于 `dropped_items`;不得再将总数计为另一种省略。 +成功截断记录在选中条目与 Receipt 上,不计为省略候选。 + +`not_retrieved` 是每个请求 family 的空集标志,`count=1`,包括 reader/head 缺失。该 family 为最终构建产生过候选时, +即使全部候选后来被丢掉,也不出现该组。其他原因统计保留轮次和最终构建处理的有界候选列表中的排除事件,而非存储中的 +不同 Artifact 数。重复身份被排除时增加 `duplicate`。放弃的扩展候选池不增加省略计数。现有 omission 和 recall-effort +计数表达同一事件时直接复用;其余排除在遍历同一有界候选池时计数,不增加数据库搜索或整份 Artifact 扫描。 + +`StageTiming` 包含 `stage`、`duration_ms` 和 `count`。对同一现有 Runtime stage 的多次执行累加耗时并记录次数; +其总和不必等于 prepare 的 wall-clock latency。只包含实际执行的阶段,包括放弃的轮次。不复制 span payload, +也不增加每个 Scope 的耗时列表。 ## 边界 | 限制 | 取值 | | ---: | ---: | -| `selected` | schema 16 条;当前 Builder 输出最多 8 条 | -| `omitted` 分组 | 16 | -| `stages` | 8 | +| `selected` | 32 条 | +| `retrieval.rounds` | 3 | +| `retrieval.searches` | 合计 32 组 | +| `retrieval.rerank_configs` | 8 个不同配置 | +| `retrieval.reranks` | 合计 32 组 | +| `omitted` | 16 组 | +| `stages` | 8 个阶段耗时聚合 | | `receipt` JSON UTF-8 大小 | 8192 bytes | -| 条目正文、query 原文、prompt、向量、密钥、token、绝对路径 | 禁止 | +| 条目正文、原始或扩展 query 原文、prompt 正文、模型响应、向量、密钥、token、绝对路径 | 禁止 | -若一份合法 Receipt 将超过 8192 bytes,Server 丢弃 Receipt,而不是截断选中身份。截断后的身份列表会与注入字节不一致。 +在实际返回给调用方的 Receipt 序列化字节上检查预算。数量上限不保证 32 个完整身份加配置证据能放入 8192 bytes。 +任何数量超限、byte 超限、身份不完整或执行证据缺失都省略整份 Receipt,记录无正文诊断,保持注入不变。 +不得截断身份、用通用 policy ID 代替配置证据,或忽略额外调用。实现验收必须用实际身份长度,测量有代表性的混合 family、 +跨 Scope 和 rerank 配置载荷。普通工作负载若频繁超出预算,应在发布前依据测量结果调整预算,不能悄悄发布持续缺失的诊断。 ## 持久化与 MCP @@ -243,16 +325,34 @@ Receipt,在同一改动里更新校验器并设置 `include_receipt`。 ## 实现要点 -1. 在 OpenAPI 中扩展 `PrepareContextRequest` 和 `PreparedContext`,再生成绑定。 -2. 继续以 `PreparedContextBuilder.build_result()` 作为选中条目来源。在 Builder 已遍历的同一候选列表上统计省略身份。 -3. 从 `ScopedContextApplication._prepare` 已得到的 Memory 搜索结果复制检索模式和 rerank 摘要。 -4. 在 Builder 返回后对 Memory 搜索已规范化的 query 和注入 `content` 做哈希。 -5. Receipt 校验失败时,记录无正文错误,并返回没有 `receipt` 键的四字段 PreparedContext。 -6. 官方宿主召回请求保持不变。只有在该宿主选择加入 Receipt 的同一改动里,才更新其校验器。 +按 Tracking Issue 的要求,仅在设计被接受后开始实现。 -聚焦测试覆盖 empty、ready、truncated、deduplicated、reranked、fallback、`include_receipt=false`(没有 `receipt` -键)、`memory_not_retrieved` / `experience_not_retrieved` 标志,以及 Receipt 组装失败。每个 ready 用例断言 -`content_digest` 与返回的 `content` 一致,且选中引用与 `origins` 一致。宿主召回夹具继续断言响应恰好四个键。 +1. 在 OpenAPI 中扩展 `PrepareContextRequest` 和 `PreparedContext`,再生成绑定。 +2. 从 `build_scopes_result()` 和代码组装保留有序选中身份及逐条渲染元数据。在同一有界候选池统计排除事件,复用已有 + budget-omission 与 recall-effort 计数。 +3. 在 `_recall_scope` 和 `_recall_round` 丢弃元数据之前,从每次搜索和推理调用保留请求内的执行证据。按轮次聚合实际 + 模式和回退原因,保留失败扩展结果。rerank 配置证据绑定到调用时的有效模型参数与实际解析 Prompt。 +4. Builder 返回后对规范化原始 query 和最终注入 `content` 做哈希。校验有序完整 origins、rerank 组与配置的引用关系、 + 所有数量上限及实际序列化 byte 大小。 +5. Receipt 任一校验失败时,记录无正文诊断,返回未改变的 PreparedContext,并省略 `receipt`。 +6. 官方宿主召回请求保持不变。只有在宿主选择加入 Receipt 的同一改动里才更新校验器。不增加 Receipt 表或渐进内容缓存。 + +实现验收覆盖: + +- empty、ready、truncated、deduplicated、reranked、fallback 和 Receipt 构造失败;默认、false、null 标志仍返回 + 恰好四个响应键。 +- 默认召回的 Topic Memory;选中 18 和 26 条历史内容的 Memory/Topic Memory/Experience/Profile 混合 assembly; + 纯代码与混合代码结果,包括代码截断后的最终行范围和哈希。 +- 两个 origins 集合的有序完整选中身份、不同 Scope 中相同简短 ID、精确读取复用,以及 `content_digest` 等于返回 + `content` 的 UTF-8 哈希。 +- 同一轮不同 Scope 中的 hybrid 与 FTS、已完成零命中与未运行的区别、正常 FTS 与推理回退、FTS 回退复用,以及扩展 + 失败时仅返回 round-zero 候选池。 +- 继承与独立 rerank 模型/参数、不同 Scope Prompt 与 revision、参数或编译 Prompt 变化导致配置 digest 变化、重复 + 配置去重、temperature 为零仍非确定性,以及注入 reranker 缺证据时 fail-open 省略 Receipt。 +- 两种预算丢弃子计数、五类来源的空集标志、放弃轮次不重复计数、每个数量上限、有代表性的序列化大小,以及 byte 超限 + 时整份省略而不丢失身份。 +- SQLite 与 OceanBase 的 Receipt 语义一致,无 schema migration,`include_receipt` 不增加召回/模型调用,成功 Receipt + 与失败诊断均无禁止内容。 # Drawbacks @@ -287,7 +387,7 @@ PreparedContext 选择发生在 Memory 搜索和 Experience 搜索之后,并 PowerContext 已有四种相关但不同的记录: -- `PreparedContextBuild.origins` 是精确选中集合,在 HTTP 边界被丢弃。 +- `PreparedContextBuild.origins` 与 `code_origins` 保留精确选中身份,在 HTTP 边界被丢弃。 - Runtime 的 `memory.search`、`experience.search`、`context.build` span 记录计数和模式,不记录引用。 - `MemoryRerankTrace` 解释进程内的 listwise Memory 搜索。 - Handoff Receipt 在 Work Continuity 中确认精确 Handoff Revision。 From 6da77b1ee6fbd129647129330b0ed01df29bdc49 Mon Sep 17 00:00:00 2001 From: TsukiKage Date: Mon, 5 Oct 2026 23:18:42 +0800 Subject: [PATCH 4/4] fix(usage): retry native SQLite deadlines before commit --- .../builtin/persistence/database.py | 41 +++++++---- .../runtime/test_model_usage_recorder.py | 69 +++++++++++++++++++ 2 files changed, 98 insertions(+), 12 deletions(-) diff --git a/src/powercontext/builtin/persistence/database.py b/src/powercontext/builtin/persistence/database.py index 04fba4dff4..ddc8a730e5 100644 --- a/src/powercontext/builtin/persistence/database.py +++ b/src/powercontext/builtin/persistence/database.py @@ -20,6 +20,7 @@ import time from collections.abc import AsyncIterator from contextlib import AbstractAsyncContextManager, asynccontextmanager, nullcontext +from sqlite3 import SQLITE_INTERRUPT from aiosqlite import Connection as SQLiteConnection from sqlalchemy.exc import OperationalError @@ -270,6 +271,24 @@ def _past_deadline() -> int: # without queuing work. Use that same thread-safe primitive in a timer so # each record needs no additional asyncio task. timer = loop.call_at(deadline, driver._conn.interrupt) + async with _sqlite_usage_attempt(connection, deadline): + yield + # busy_timeout stays bounded through COMMIT and any automatic ROLLBACK. + finally: + if timer is not None: + timer.cancel() + await _restore_sqlite_usage_connection(driver, busy_timeout) + # SQLite cannot interrupt a user-defined function or a blocked filesystem + # syscall. Await native cleanup rather than falsely report it as cancelled. + + +@asynccontextmanager +async def _sqlite_usage_attempt(connection: AsyncConnection, deadline: float) -> AsyncIterator[None]: + """Repeat only an expired body, never an interruption with an unknown commit.""" + + loop = asyncio.get_running_loop() + committing = False + try: async with connection.begin(): if loop.time() >= deadline: raise ModelUsageAttemptExpired @@ -277,20 +296,18 @@ def _past_deadline() -> int: # Include the Scope existence read in the write's actual snapshot. await connection.exec_driver_sql("BEGIN") yield - # The body is done, but the attempt may have outlived its slice. The - # statements are still uncommitted, so stopping here applies nothing - # and costs the record its usage; the native timer above, not this - # check, is what bounds work in progress. Expire as a repeatable - # attempt and let the recorder spend the rest of the record budget. + # Native interruption and a body that returns after its slice are + # both unapplied attempts. Let the recorder spend its remaining budget. if loop.time() >= deadline: raise ModelUsageAttemptExpired - # busy_timeout stays bounded through COMMIT and any automatic ROLLBACK. - finally: - if timer is not None: - timer.cancel() - await _restore_sqlite_usage_connection(driver, busy_timeout) - # SQLite cannot interrupt a user-defined function or a blocked filesystem - # syscall. Await native cleanup rather than falsely report it as cancelled. + committing = True + except OperationalError as error: + # The transaction has rolled back. Connection-level restoration must + # also succeed before the enclosing context can return this expiry. + sqlite_code = getattr(error.orig, "sqlite_errorcode", None) + if not committing and sqlite_code == SQLITE_INTERRUPT and loop.time() >= deadline: + raise ModelUsageAttemptExpired from error + raise def _disabled_progress() -> int: diff --git a/tests/builtin/runtime/test_model_usage_recorder.py b/tests/builtin/runtime/test_model_usage_recorder.py index ebb0a797b5..b9c3e8ecf0 100644 --- a/tests/builtin/runtime/test_model_usage_recorder.py +++ b/tests/builtin/runtime/test_model_usage_recorder.py @@ -696,6 +696,43 @@ async def record( await super().record(connection, scope_id, usage_date, purpose, operation, usage) +@pytest.mark.parametrize("file_backed", [False, True]) +def test_native_deadline_before_commit_retries_usage_without_double_counting(tmp_path: Path, file_backed: bool) -> None: + class InterruptedOnceRepository(StatisticsRepository): + slow = True + + async def record(self, connection, *args) -> None: + # Apply the increment before real SQLite VM work consumes the first + # attempt's slice. Its rollback must remove that increment before a + # retry, rather than either dropping the record or counting it twice. + await super().record(connection, *args) + if self.slow: + self.slow = False + await connection.exec_driver_sql( + "WITH RECURSIVE n(x) AS (VALUES(1) UNION ALL SELECT x+1 FROM n WHERE x<1000000000) SELECT sum(x) FROM n" + ) + + async def scenario() -> None: + config = ( + SQLiteConfig(url=f"sqlite+aiosqlite:///{tmp_path / 'interrupted-usage.db'}") + if file_backed + else SQLiteConfig() + ) + async with _database(config) as database: + recorder = _ModelUsageRecorder(database, InterruptedOnceRepository(), write_timeout_seconds=0.4) + try: + _offer(recorder, InferenceUsage(requests=1, input_tokens=3, output_tokens=5)) + await recorder.flush() + rows = await _rows(database) + assert len(rows) == 1 + assert (rows[0].requests, rows[0].input_tokens, rows[0].output_tokens) == (1, 3, 5) + await _assert_connection_restored(database) + finally: + await recorder.close() + + asyncio.run(scenario()) + + def test_native_sqlite_deadline_stops_vm_work_and_preserves_in_memory_database() -> None: async def scenario() -> None: async with _database() as database: @@ -718,6 +755,38 @@ async def scenario() -> None: asyncio.run(scenario()) +def test_an_interrupted_commit_reply_does_not_repeat_applied_usage(caplog: pytest.LogCaptureFixture) -> None: + async def scenario() -> None: + async with _database() as database: + + def lost_reply(connection) -> None: + # Commit for real, then let the native deadline expire before + # reporting an interrupted reply. Repeating this outcome would + # increment an already committed record a second time. + connection.connection.commit() + await_only(asyncio.sleep(0.12)) + original = sqlite3.OperationalError("secret commit reply") + original.sqlite_errorcode = sqlite3.SQLITE_INTERRUPT + raise OperationalError("COMMIT", None, original) + + event.listen(database.engine.sync_engine, "commit", lost_reply, once=True) + recorder = _ModelUsageRecorder(database, StatisticsRepository(), write_timeout_seconds=0.4) + try: + _offer(recorder) + await recorder.flush() + rows = await _rows(database) + assert len(rows) == 1 + assert rows[0].requests == 1 + await _assert_connection_restored(database) + finally: + await recorder.close() + + with caplog.at_level(logging.WARNING): + asyncio.run(scenario()) + assert "will not be retried" in caplog.text + assert "secret commit reply" not in caplog.text + + class _BlockingNativeRepository(StatisticsRepository): def __init__(self) -> None: self.entered = ThreadEvent()