Compare commits
21 Commits
ef19a03cbf
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
| 916532100d | |||
| bef8ca263b | |||
| f6d037b613 | |||
| 449b506804 | |||
| 071a78ae22 | |||
| ad1cba41c2 | |||
| 67338798aa | |||
| e95e2f2220 | |||
| 628b8d1800 | |||
| d24d4609b6 | |||
| c7f79fb38e | |||
| 8c071800cf | |||
| 569e12c6f4 | |||
| 5b6eb591b2 | |||
| b630384aa0 | |||
| 297d58f090 | |||
| ee71dc4937 | |||
| 65e5d65d14 | |||
| b40b40aaf3 | |||
| f120be1cb4 | |||
| 17673d74ea |
@@ -39,6 +39,8 @@ demonstrates all implemented D&D lanes and the supporting campaign references.
|
||||
artifact formats.
|
||||
- [Subprocess consumer guide](docs/consumers/subprocess.md) — invoke Notarius
|
||||
from an orchestrator and consume a published result.
|
||||
- [Complete D&D consumer guide](docs/consumers/dnd-pipeline.md) — run the full
|
||||
D&D pipeline as a subprocess and discover its structured artifacts.
|
||||
- [Internal overview](docs/internal/overview.md) — implemented component map
|
||||
for maintainers.
|
||||
- [Developer guide](docs/development.md) — contributor orientation and
|
||||
|
||||
@@ -1,47 +0,0 @@
|
||||
{
|
||||
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
||||
"$id": "notarius.dnd.entity_reconcile.llm",
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["duplicate_groups"],
|
||||
"properties": {
|
||||
"duplicate_groups": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["members", "canonical"],
|
||||
"properties": {
|
||||
"members": {
|
||||
"type": "array",
|
||||
"items": {"$ref": "#/$defs/selector"}
|
||||
},
|
||||
"canonical": {"$ref": "#/$defs/selector"}
|
||||
}
|
||||
}
|
||||
}
|
||||
},
|
||||
"$defs": {
|
||||
"selector": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["name", "source_refs"],
|
||||
"properties": {
|
||||
"name": {"type": "string", "minLength": 1},
|
||||
"source_refs": {
|
||||
"type": "array",
|
||||
"minItems": 1,
|
||||
"items": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["start_unit_id", "end_unit_id"],
|
||||
"properties": {
|
||||
"start_unit_id": {"type": "integer", "minimum": 1},
|
||||
"end_unit_id": {"type": "integer", "minimum": 1}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,8 +1,9 @@
|
||||
Use candidate names and cited transcript windows only to determine whether
|
||||
candidates identify the same item type or unique designation. Do not treat
|
||||
nearby evidence, similar objects, or a shared owner as sufficient. Keep
|
||||
currency denominations, materially different item types, and uncertain aliases
|
||||
separate. Do not infer an item property or uniqueness.
|
||||
Determine whether candidates identify the same item type or unique designation
|
||||
using their contextual labels and cited transcript windows. Do not treat nearby
|
||||
evidence, similar objects, or a shared owner as sufficient.
|
||||
|
||||
Keep currency denominations and materially different item types separate. Keep
|
||||
uncertain aliases separate. Do not infer an item property or uniqueness.
|
||||
|
||||
When selecting a canonical display name, choose one supplied candidate name
|
||||
that is the clearest established designation.
|
||||
|
||||
@@ -12,19 +12,19 @@ messages:
|
||||
- role: system
|
||||
content_file: ./sharedassets/common-dnd-system.md
|
||||
- role: user
|
||||
content_file: ./instructions.md
|
||||
content_file: ./sharedassets/protocol.md
|
||||
- role: user
|
||||
content_file: ./sharedassets/common-dnd-entity-reconciliation.md
|
||||
content_file: ./instructions.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
- role: user
|
||||
content_file: ./candidates.md
|
||||
content_file: ./sharedassets/candidates.md
|
||||
- role: user
|
||||
content_file: ./sharedassets/common-dnd-transcript-windows.md
|
||||
content_file: ./sharedassets/transcript-windows.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_entity_reconcile_llm.v1.json
|
||||
schema_path: semantic_reconciliation_llm.v1.json
|
||||
repair_attempts: 0
|
||||
|
||||
@@ -1,2 +0,0 @@
|
||||
Location candidates:
|
||||
{{ input "candidates" }}
|
||||
@@ -1,6 +1,8 @@
|
||||
Use candidate names and their cited transcript windows to determine whether
|
||||
candidates identify the same physical place. Do not treat matching names,
|
||||
nearby evidence, nested places, or generic labels as sufficient. Keep parent
|
||||
and child places, similarly named places, and uncertain aliases separate.
|
||||
Determine whether candidates identify the same physical place using their
|
||||
contextual labels and cited transcript windows. Do not treat matching names,
|
||||
nearby evidence, nested places, or generic labels as sufficient.
|
||||
|
||||
Keep parent and child places separate, as well as similarly named places and
|
||||
uncertain aliases.
|
||||
|
||||
When selecting a canonical display name, prefer the clearest established name.
|
||||
|
||||
@@ -12,19 +12,19 @@ messages:
|
||||
- role: system
|
||||
content_file: ./sharedassets/common-dnd-system.md
|
||||
- role: user
|
||||
content_file: ./instructions.md
|
||||
content_file: ./sharedassets/protocol.md
|
||||
- role: user
|
||||
content_file: ./sharedassets/common-dnd-entity-reconciliation.md
|
||||
content_file: ./instructions.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
- role: user
|
||||
content_file: ./candidates.md
|
||||
content_file: ./sharedassets/candidates.md
|
||||
- role: user
|
||||
content_file: ./sharedassets/common-dnd-transcript-windows.md
|
||||
content_file: ./sharedassets/transcript-windows.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_entity_reconcile_llm.v1.json
|
||||
schema_path: semantic_reconciliation_llm.v1.json
|
||||
repair_attempts: 0
|
||||
|
||||
@@ -1,3 +0,0 @@
|
||||
NPC candidates for identity comparison:
|
||||
|
||||
{{ input "candidates" }}
|
||||
@@ -1,6 +1,7 @@
|
||||
Use candidate aliases and their cited transcript windows to determine whether
|
||||
candidates refer to the same individual. Preserve distinct individuals even
|
||||
when their names are similar.
|
||||
Determine whether candidates refer to the same individual using their
|
||||
contextual labels and cited transcript windows. Preserve distinct individuals
|
||||
even when their names are similar or their contextual descriptions are
|
||||
identical.
|
||||
|
||||
When selecting a canonical display name, prefer a complete, stable proper name
|
||||
over an abbreviation. Prefer an unadorned proper name over that name plus a
|
||||
|
||||
@@ -12,19 +12,19 @@ messages:
|
||||
- role: system
|
||||
content_file: ./sharedassets/common-dnd-system.md
|
||||
- role: user
|
||||
content_file: ./instructions.md
|
||||
content_file: ./sharedassets/protocol.md
|
||||
- role: user
|
||||
content_file: ./sharedassets/common-dnd-entity-reconciliation.md
|
||||
content_file: ./instructions.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
- role: user
|
||||
content_file: ./candidates.md
|
||||
content_file: ./sharedassets/candidates.md
|
||||
- role: user
|
||||
content_file: ./sharedassets/common-dnd-transcript-windows.md
|
||||
content_file: ./sharedassets/transcript-windows.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: dnd_entity_reconcile_llm.v1.json
|
||||
schema_path: semantic_reconciliation_llm.v1.json
|
||||
repair_attempts: 0
|
||||
|
||||
@@ -1,7 +0,0 @@
|
||||
Identify only well-supported duplicate groups among the supplied candidates.
|
||||
|
||||
Return each selected candidate's supplied contextual descriptor exactly: its
|
||||
`name` and complete ordered `source_refs`. A group must contain at least two
|
||||
supplied descriptors, and its `canonical` descriptor must be one of its
|
||||
members. Do not invent names, ranges, records, evidence, or replacement values.
|
||||
Omit any uncertain or unsafe group.
|
||||
@@ -1,7 +1,6 @@
|
||||
Transcript units are the only evidence for extracted events and factual claims.
|
||||
Every reported factual claim must be supported by cited transcript units. Use
|
||||
integer `start_unit_id` and `end_unit_id` values from the transcript. Omit
|
||||
`source_id`; Notarius assigns the current source identity.
|
||||
integer `start_unit_id` and `end_unit_id` values from the transcript.
|
||||
|
||||
When supporting evidence is non-contiguous, use multiple narrow ranges rather
|
||||
than a broad range that bridges unrelated conversation.
|
||||
|
||||
@@ -1,7 +1,5 @@
|
||||
You process Dungeons & Dragons gameplay transcripts.
|
||||
|
||||
Rely only on the supplied inputs. They may contain transcription errors,
|
||||
repeated lines, incomplete sentences, and misheard proper nouns.
|
||||
As input, you will receive one or more portions of a transcript. The transcript may contain transcription errors, repeated lines, incomplete sentences, and misheard proper nouns.
|
||||
|
||||
Return exactly one JSON object that conforms to the configured response schema,
|
||||
with no explanatory prose.
|
||||
Return exactly one JSON object that conforms to the configured response schema, with no explanatory prose.
|
||||
|
||||
@@ -1,5 +1,3 @@
|
||||
One extraction chunk from a Dungeons & Dragons gameplay transcript is provided
|
||||
below. Report and infer only what is within this chunk. Its unit IDs retain
|
||||
their source-wide meaning.
|
||||
One extraction chunk from a Dungeons & Dragons gameplay transcript is provided below. Report and infer only what is within this chunk. Its unit IDs retain their source-wide meaning.
|
||||
|
||||
{{ input "transcript" }}
|
||||
|
||||
@@ -1,4 +1,3 @@
|
||||
The complete ordered transcript of this Dungeons & Dragons gameplay session is
|
||||
provided below. It may contain multiple scenes.
|
||||
The complete ordered transcript of this Dungeons & Dragons gameplay session is provided below.
|
||||
|
||||
{{ input "transcript" }}
|
||||
|
||||
@@ -1,6 +0,0 @@
|
||||
Selected Dungeons & Dragons gameplay transcript evidence windows are provided
|
||||
below. They may be incomplete, non-contiguous, or overlapping. Use them to
|
||||
evaluate candidate identity, but do not treat absence outside these windows as
|
||||
evidence.
|
||||
|
||||
{{ input "transcript" }}
|
||||
@@ -1,2 +1,3 @@
|
||||
Item candidates:
|
||||
Candidate material:
|
||||
|
||||
{{ input "candidates" }}
|
||||
@@ -0,0 +1,5 @@
|
||||
Identify only high-confidence duplicate entities among the supplied candidates.
|
||||
|
||||
Preserve distinct entities even when their names are similar. Treat contextual descriptions and transcript evidence as supporting material, not as permission to merge ambiguous records.
|
||||
|
||||
When several records are duplicates, choose as canonical the candidate with the clearest stable identity. Prefer a complete proper name over an abbreviation, and prefer an unadorned proper name over one with incidental descriptors unless the evidence establishes those descriptors as part of the name. A longer name is not inherently more canonical.
|
||||
27
assets/generic/normalize/deduplication/prompts/prompt.yaml
Normal file
27
assets/generic/normalize/deduplication/prompts/prompt.yaml
Normal file
@@ -0,0 +1,27 @@
|
||||
id: generic.semantic_reconciliation
|
||||
version: "v1"
|
||||
inputs:
|
||||
- name: candidates
|
||||
required: true
|
||||
content_type: application/json
|
||||
- name: transcript
|
||||
required: true
|
||||
content_type: application/json
|
||||
messages:
|
||||
- role: system
|
||||
content_file: ./system.md
|
||||
- role: user
|
||||
content_file: ./protocol.md
|
||||
- role: user
|
||||
content_file: ./instructions.md
|
||||
cache_control:
|
||||
type: ephemeral
|
||||
- role: user
|
||||
content_file: ./candidates.md
|
||||
- role: user
|
||||
content_file: ./transcript-windows.md
|
||||
output:
|
||||
format: json
|
||||
validation_mode: json_schema
|
||||
schema_path: semantic_reconciliation_llm.v1.json
|
||||
repair_attempts: 0
|
||||
@@ -0,0 +1,7 @@
|
||||
Use only the positive integer `candidate_id` values supplied in the candidate material.
|
||||
|
||||
Return a duplicate group only when the evidence supports that every selected candidate describes the same underlying entity. Each group must contain at least two distinct candidate IDs, and its `canonical_candidate_id` must be one of those IDs. A candidate may appear in at most one group.
|
||||
|
||||
Omit uncertain matches and candidates that should remain distinct. Do not invent candidates or infer an ID from list position. An empty `duplicate_groups` array is valid.
|
||||
|
||||
The response must conform exactly to the selected JSON schema. Return IDs only: do not copy candidate names, evidence, transcript text, source identifiers, or source ranges into the response.
|
||||
2
assets/generic/normalize/deduplication/prompts/system.md
Normal file
2
assets/generic/normalize/deduplication/prompts/system.md
Normal file
@@ -0,0 +1,2 @@
|
||||
You reconcile structured records that may describe the same underlying entity.
|
||||
Follow the supplied protocol and return only the requested structured result.
|
||||
@@ -0,0 +1,3 @@
|
||||
Transcript evidence windows:
|
||||
|
||||
{{ input "transcript" }}
|
||||
@@ -0,0 +1,32 @@
|
||||
{
|
||||
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
||||
"$id": "notarius.generic.semantic_reconciliation.llm",
|
||||
"title": "notarius_semantic_reconciliation_llm_v1",
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["duplicate_groups"],
|
||||
"properties": {
|
||||
"duplicate_groups": {
|
||||
"type": "array",
|
||||
"items": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["candidate_ids", "canonical_candidate_id"],
|
||||
"properties": {
|
||||
"candidate_ids": {
|
||||
"type": "array",
|
||||
"minItems": 2,
|
||||
"items": {
|
||||
"type": "integer",
|
||||
"minimum": 1
|
||||
}
|
||||
},
|
||||
"canonical_candidate_id": {
|
||||
"type": "integer",
|
||||
"minimum": 1
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,91 @@
|
||||
# ADR-0013: Use request-local candidate handles for semantic reconciliation
|
||||
|
||||
**Status:** Accepted
|
||||
**Date:** 2026-08-09
|
||||
|
||||
## Context
|
||||
|
||||
Several typed normalize stage modules need semantic reconciliation after
|
||||
deterministic preprocessing: a model can judge whether source-backed candidates
|
||||
refer to the same underlying entity, while application code remains responsible
|
||||
for constructing the normalized artifact. Requiring the model to reproduce a
|
||||
candidate's full contextual selector makes the response larger and introduces
|
||||
avoidable formatting, ordering, and transcription failure modes.
|
||||
|
||||
Reconciliation must preserve the exact typed artifact boundary established by
|
||||
[ADR-0003](0003-typed-interfaces-with-two-zone-data-model.md), the domain-neutral
|
||||
framework and concrete-domain dependency direction established by
|
||||
[ADR-0004](0004-package-modules-by-domain.md), and the distinction in
|
||||
[ADR-0009](0009-minimal-evidence-grounded-extraction-artifacts.md) between source
|
||||
evidence and auxiliary identity context. It also needs a concrete, narrowly
|
||||
scoped application of the request-local-label exception allowed by
|
||||
[ADR-0012](0012-resolve-opaque-entity-identifiers-deterministically.md).
|
||||
|
||||
## Decision
|
||||
|
||||
Semantic reconciliation will be a domain-neutral framework mechanism used by
|
||||
typed normalize stage modules. A consuming artifact family will retain
|
||||
ownership of its typed records, identity rules, consolidation policy, durable
|
||||
IDs, and domain warnings; the framework mechanism will not infer those rules
|
||||
from arbitrary data.
|
||||
|
||||
For each reconciliation request, deterministic code will assign every eligible
|
||||
model-visible candidate a contiguous, one-based integer handle. The model may
|
||||
receive the candidate's contextual label, source references, and bounded source
|
||||
context needed to judge identity, but its structured response will identify
|
||||
candidates only by those supplied handles. A handle is local to one request,
|
||||
does not represent entity identity, and must never enter a durable artifact or
|
||||
be used to derive a durable ID.
|
||||
|
||||
The model will propose duplicate groups and select one supplied member of each
|
||||
group as canonical. Deterministic code will resolve the handles through the
|
||||
retained request mapping, validate the complete proposal, discard unsafe
|
||||
groups, and apply only validated groups through typed domain-owned policy. The
|
||||
model will not synthesize replacement records or directly mutate an artifact.
|
||||
|
||||
Every reconciliation prompt will combine a mandatory framework-owned protocol
|
||||
and safety policy with an explicitly selected semantic policy. The semantic
|
||||
policy may be the conservative generic policy or a domain-owned policy, but it
|
||||
cannot replace the shared response protocol or deterministic safety boundary.
|
||||
|
||||
## Alternatives considered
|
||||
|
||||
- Return durable application IDs. Opaque IDs do not help semantic judgment,
|
||||
expose application identity mechanics, and make model output reproduce data
|
||||
that deterministic code already owns.
|
||||
- Return names alone or copied contextual selectors. Names can be ambiguous,
|
||||
while reproducing labels and source ranges adds response complexity and
|
||||
creates mismatches without adding semantic information. Request-local
|
||||
handles preserve exact selection without either failure mode.
|
||||
- Ask the model to return synthesized canonical replacement records. This
|
||||
would transfer typed artifact construction, provenance consolidation, and
|
||||
durable identity policy to a probabilistic boundary.
|
||||
- Reconcile reflection-discovered fields or arbitrary JSON. This would weaken
|
||||
the typed artifact contract and move domain semantics into generic code.
|
||||
- Hide reconciliation inside extraction or another stage. This would obscure
|
||||
stage ownership and create cross-stage behavior outside the fixed pipeline;
|
||||
reconciliation remains explicit normalize-stage behavior.
|
||||
- Let each domain replace the complete prompt protocol. This would duplicate
|
||||
safety mechanics and allow domain policy to bypass the common response and
|
||||
validation contract.
|
||||
|
||||
## Consequences
|
||||
|
||||
Model responses become smaller and easier to validate, while deterministic
|
||||
application code retains authority over identity, provenance, ordering, and
|
||||
typed artifact construction. The framework requires a request-local mapping,
|
||||
bounded context preparation, a private integer response contract, proposal
|
||||
assessment, and shared prompt assets. Each consuming artifact family still
|
||||
requires a typed adapter for its irreducibly domain-specific rules.
|
||||
|
||||
Request-local handles are deliberately unsuitable for persistence, logging as
|
||||
entity identity, checkpoint contracts, or cross-request correlation. Changes
|
||||
to shared protocol and policy assets must participate in the normal prompt,
|
||||
schema, and checkpoint fingerprint mechanisms.
|
||||
|
||||
Acceptance of this decision does not imply that the shared mechanism or its
|
||||
consumer migrations are implemented. The
|
||||
[feature roadmap](../roadmap/semantic-reconciliation.md) owns target behavior
|
||||
and status, and the
|
||||
[implementation plan](../roadmap/implementation.md) owns delivery sequence
|
||||
until the work is complete.
|
||||
@@ -319,7 +319,8 @@ Unknown outer or nested option fields are rejected, as are incompatible YAML
|
||||
types. The allowlist remains valid when a run uses lane filtering: a configured
|
||||
lane that is not active for that invocation simply contributes no evidence.
|
||||
Evidence publication is opt-in because it can persist source text and metadata.
|
||||
Its payload contract is [Published Evidence Context](integrations/evidence-context.md).
|
||||
When enabled, it publishes the selected source-unit excerpt defined by the
|
||||
[Published Evidence Context contract](integrations/evidence-context.md).
|
||||
|
||||
## References And Ordered Handoffs
|
||||
|
||||
|
||||
202
docs/consumers/dnd-pipeline.md
Normal file
202
docs/consumers/dnd-pipeline.md
Normal file
@@ -0,0 +1,202 @@
|
||||
# Consuming The Complete D&D Pipeline
|
||||
|
||||
Use this workflow when an orchestrator runs the maintained complete D&D
|
||||
pipeline and consumes its structured JSON artifacts. The generic
|
||||
[subprocess consumer guide](subprocess.md) owns process-level responsibilities;
|
||||
this guide connects that workflow to the complete D&D configuration, its
|
||||
Seriatim input, and its artifact inventory.
|
||||
|
||||
The [CLI reference](../cli.md), [configuration reference](../config.md),
|
||||
[run-result receipt](../integrations/run-result.md), and
|
||||
[published JSON output contract](../integrations/json-output.md) remain the
|
||||
canonical definitions of those public interfaces.
|
||||
|
||||
## Prepare And Validate The Deployment
|
||||
|
||||
Start from the maintained
|
||||
[complete D&D configuration](../../examples/dnd-complete.config.yml). It uses
|
||||
the `dnd-session` pipeline and demonstrates every implemented D&D lane, ordered
|
||||
artifact handoffs, campaign references, chunk-map publication, and evidence
|
||||
context.
|
||||
|
||||
A deployment must provide its own PromptKit profile and campaign reference
|
||||
files. Use absolute paths for service and subprocess deployments. In
|
||||
particular, observe these different resolution rules:
|
||||
|
||||
- reference paths in YAML are resolved relative to the Notarius configuration
|
||||
file; and
|
||||
- `promptkit.profile_file` is resolved relative to the Notarius process working
|
||||
directory.
|
||||
|
||||
Do not copy the repository example's relative profile path into a deployment
|
||||
without also controlling that working directory. The complete path and profile
|
||||
rules are defined in [Configuration](../config.md).
|
||||
|
||||
Preflight the deployed configuration before processing sessions and whenever
|
||||
it changes:
|
||||
|
||||
```sh
|
||||
notarius config validate \
|
||||
--config /absolute/path/to/notarius.yml \
|
||||
--pipeline dnd-session
|
||||
```
|
||||
|
||||
Provide credentials through the environment or the documented configuration
|
||||
mechanism. Do not put credentials in command arguments, generated
|
||||
configuration, or logs.
|
||||
|
||||
## Supply The Transcript
|
||||
|
||||
The complete pipeline consumes a Seriatim JSON document. The
|
||||
[Seriatim input contract](../integrations/seriatim.md) defines its required
|
||||
metadata, segments, and validation rules. Preserve segment IDs: D&D artifact
|
||||
citations use those segment IDs as source-unit ranges.
|
||||
|
||||
When the caller maintains several transcript tiers, use the final trimmed JSON
|
||||
transcript so extraction operates on the same session content presented to
|
||||
later consumers. For example, Narratio identifies this implemented artifact as
|
||||
`narratio.transcript.final_trimmed` and normally stores it at
|
||||
`transcripts/final.trimmed.json`.
|
||||
|
||||
Notarius generates a stable prompt session from the resolved input module and
|
||||
the exact input bytes. An ordinary orchestrator should not pass `--session-id`.
|
||||
Use that override only when intentionally changing the routing relationship
|
||||
between invocations; it is not a credential or output identity.
|
||||
|
||||
## Run Notarius
|
||||
|
||||
Invoke the pipeline with explicit absolute paths and request its
|
||||
machine-readable receipt:
|
||||
|
||||
```sh
|
||||
notarius run dnd-session \
|
||||
--config /absolute/path/to/notarius.yml \
|
||||
--input /absolute/path/to/transcripts/final.trimmed.json \
|
||||
--output-dir /absolute/path/to/notarius-output \
|
||||
--json
|
||||
```
|
||||
|
||||
The caller should:
|
||||
|
||||
- capture stdout and stderr separately;
|
||||
- propagate cancellation and impose an operator-appropriate timeout;
|
||||
- wait for process completion before interpreting stdout; and
|
||||
- retain stderr for diagnosis without copying secrets or transcript content
|
||||
into other logs.
|
||||
|
||||
Only exit status 0 permits decoding stdout as a receipt. Ignore stdout after a
|
||||
nonzero exit because a failed receipt write can leave partial bytes. The
|
||||
[CLI reference](../cli.md#output-streams-and-exit-statuses) defines the complete
|
||||
stream and exit-status contract.
|
||||
|
||||
## Discover The Published Bundle
|
||||
|
||||
Decode the successful stdout document as a supported run-result schema. For
|
||||
the current contract, `schema_version` is `notarius.run-result.v1`. Tolerate
|
||||
unknown fields allowed by that version, but reject an unsupported schema
|
||||
version.
|
||||
|
||||
Use the receipt's absolute `output_directory` as the exact run-specific bundle
|
||||
root. Do not scan the output root for its newest directory, guess a run ID, or
|
||||
construct a bundle path. Resolve `index_file` beneath `output_directory` and
|
||||
reject an absolute logical path or any result that escapes the bundle root.
|
||||
|
||||
Read `index.json` and locate each requested lane in `output_files` by its exact
|
||||
`lane_id`. Do not guess a lane filename. Before decoding a payload:
|
||||
|
||||
1. resolve its descriptor's relative `file` beneath the bundle root with the
|
||||
same confinement check;
|
||||
2. verify the descriptor's media type and schema identity against the linked
|
||||
artifact contract; and
|
||||
3. decode the payload according to that contract.
|
||||
|
||||
The [published JSON output contract](../integrations/json-output.md) defines
|
||||
the index and bundle layout. Treat all paths obtained from a decoded external
|
||||
document as untrusted until confined to their documented root.
|
||||
|
||||
## Complete Artifact Inventory
|
||||
|
||||
When every configured lane is accepted, the complete example publishes these
|
||||
lane artifacts:
|
||||
|
||||
| Lane ID | Purpose | Canonical contract |
|
||||
| --- | --- | --- |
|
||||
| `item-registry` | Canonical registry of encountered items and currency. | [Item registry](../integrations/dnd-item-registry-artifacts.md) |
|
||||
| `npc-registry` | Canonical registry of named NPCs. | [NPC registry](../integrations/dnd-npc-registry-artifacts.md) |
|
||||
| `location-registry` | Canonical registry of named locations. | [Location registry](../integrations/dnd-location-registry-artifacts.md) |
|
||||
| `scene-descriptions` | Classification, title, and summary for each scene. | [Scene descriptions](../integrations/dnd-scene-description-artifacts.md) |
|
||||
| `item-occurrences` | Source-grounded item discovery, acquisition, use, transfer, and loss events. | [Item occurrences](../integrations/dnd-item-occurrence-artifacts.md) |
|
||||
| `spells` | Source-grounded spell casts and casters. | [Spell casts](../integrations/dnd-spell-artifacts.md) |
|
||||
| `combat-turns` | Source-grounded combat turn participation. | [Combat turns](../integrations/dnd-combat-turn-artifacts.md) |
|
||||
| `npc-occurrences` | Source-grounded NPC interaction occurrences. | [NPC occurrences](../integrations/dnd-npc-occurrence-artifacts.md) |
|
||||
| `location-occurrences` | Source-grounded location occurrences. | [Location occurrences](../integrations/dnd-location-occurrence-artifacts.md) |
|
||||
| `enemy-events` | Source-grounded enemy combat events. | [Enemy events](../integrations/dnd-enemy-event-artifacts.md) |
|
||||
|
||||
The JSON encoder always publishes these bundle-management files:
|
||||
|
||||
| File | Purpose |
|
||||
| --- | --- |
|
||||
| `index.json` | Discovery document for lane and pipeline-wide artifacts. |
|
||||
| `manifest.json` | Run provenance and result summaries. |
|
||||
| `rejected.json` | Rejected pipeline outputs. |
|
||||
| `warnings.json` | Accepted-output and run warnings. |
|
||||
|
||||
The complete configuration also requests two pipeline-wide artifacts:
|
||||
|
||||
- [`chunk-map.json`](../integrations/chunk-map.md), the accepted chunk plan and
|
||||
chunk metadata; and
|
||||
- [`evidence-context.json`](../integrations/evidence-context.md), a reading
|
||||
excerpt containing the union of selected cited source units and the
|
||||
configured surrounding window.
|
||||
|
||||
Discover both from their top-level `index.json` descriptors rather than
|
||||
treating them as lanes. Evidence context is convenient reading material, not
|
||||
authoritative provenance; citations in the normalized lane payloads remain the
|
||||
evidence contract.
|
||||
|
||||
Every optional or lane file is published only when its corresponding artifact
|
||||
is available. A successful process does not guarantee that all configured
|
||||
lanes were accepted.
|
||||
|
||||
## Decide What Counts As Consumer Success
|
||||
|
||||
Exit status 0 means Notarius completed the pipeline and published its result
|
||||
bundle. The receipt or bundle may still report warnings, rejected outputs, or
|
||||
missing lane descriptors. A downstream consumer must define its own required
|
||||
artifact set explicitly.
|
||||
|
||||
A caller that claims to consume the complete D&D workflow should normally
|
||||
require all ten lane IDs in the table and verify each descriptor's expected
|
||||
contract. If any required lane is missing, rejected, or incompatible, fail the
|
||||
caller's extraction step while retaining the Notarius bundle for diagnosis. A
|
||||
consumer that needs only a subset may define and document a narrower policy.
|
||||
|
||||
Keep the successful receipt with the complete published bundle. Retain
|
||||
`manifest.json`, `rejected.json`, `warnings.json`, and captured process logs as
|
||||
required by the caller's provenance, diagnosis, and retention policies. Avoid
|
||||
selectively copying payload files without also preserving enough index and
|
||||
manifest information to identify their originating run and contracts.
|
||||
|
||||
The transcript, lane artifacts, evidence context, manifest, debug data, and
|
||||
logs can all contain private campaign information. Apply the same access,
|
||||
publication, and retention controls used for the source transcript.
|
||||
|
||||
## Consumer Checklist
|
||||
|
||||
- Validate the deployed Notarius configuration and `dnd-session` pipeline.
|
||||
- Pass the final trimmed Seriatim JSON transcript with stable segment IDs.
|
||||
- Use absolute configuration, input, output-root, profile, and reference paths
|
||||
in service deployments.
|
||||
- Capture stdout and stderr separately and enforce cancellation and timeout.
|
||||
- Parse stdout only after exit status 0.
|
||||
- Accept only supported receipt, index, and artifact schema versions while
|
||||
tolerating permitted unknown fields.
|
||||
- Use the receipt's `output_directory`; never guess the run directory.
|
||||
- Confine `index_file` and every descriptor path to the published bundle root.
|
||||
- Discover lanes by `lane_id` and verify descriptor compatibility before
|
||||
decoding payloads.
|
||||
- Enforce an explicit required-lane policy and inspect rejections and warnings.
|
||||
- Preserve the receipt and sufficient bundle provenance for every retained
|
||||
artifact.
|
||||
- Protect all transcript-derived files and diagnostic streams as sensitive
|
||||
campaign data.
|
||||
@@ -6,6 +6,10 @@ statuses, while the [run-result receipt](../integrations/run-result.md) and
|
||||
[Published JSON Output contract](../integrations/json-output.md) own the
|
||||
durable result formats.
|
||||
|
||||
For the maintained complete D&D workflow, including its transcript input,
|
||||
configured lane inventory, and downstream acceptance checklist, see
|
||||
[Consuming The Complete D&D Pipeline](dnd-pipeline.md).
|
||||
|
||||
## Run And Check The Process
|
||||
|
||||
Optionally preflight a selected configuration and pipeline before work starts:
|
||||
@@ -55,10 +59,10 @@ contract. The JSON bundle contract links to the available lane contracts.
|
||||
If `index.json` has an `evidence_context` descriptor, treat it as a
|
||||
pipeline-wide artifact rather than a lane entry. Verify its six descriptor
|
||||
fields before decoding the linked file according to the [Published Evidence
|
||||
Context contract](../integrations/evidence-context.md). Use each
|
||||
`evidence_refs` entry as the citation to source material. Its surrounding
|
||||
context range and included units explain the citation, but do not widen or
|
||||
replace the cited source reference.
|
||||
Context contract](../integrations/evidence-context.md). Decode its top-level
|
||||
source-unit array as a reading excerpt. Obtain authoritative citations and lane
|
||||
provenance from the normalized lane artifacts; the excerpt has neither and its
|
||||
nearby units do not widen a lane artifact's cited source reference.
|
||||
|
||||
A zero exit status may still report rejected outputs, warnings, or absent
|
||||
lanes. The caller decides which lane IDs are required for its own work and
|
||||
@@ -73,5 +77,5 @@ them. Treat the input, output bundle, cache, debug bundle, and captured process
|
||||
logs as potentially sensitive data. Apply the caller's access controls and
|
||||
retention policy, and avoid copying secrets into arguments, logs, or
|
||||
provenance records. An evidence-context artifact contains source-unit text and
|
||||
metadata, and selected lanes can cover most of an input; preserve and share it
|
||||
only when that source content is authorized for the recipient.
|
||||
metadata and can cover most of an input; preserve and share it only when that
|
||||
source content is authorized for the recipient.
|
||||
|
||||
@@ -18,7 +18,7 @@ implemented component map.
|
||||
| Any documentation addition or revision | [Documentation Policy](policy/documentation.md) | It defines canonical homes, audiences, current-behavior rules, and maintenance requirements. |
|
||||
| Adding, changing, reviewing, or deleting tests | [Testing Policy](policy/testing.md) | It defines risk-based sufficiency, durable test boundaries, test-double guidance, and criteria for retaining tests. |
|
||||
| CLI composition or command behavior | [CLI Internals](internal/cli.md) and [CLI Reference](cli.md) | The internal guide owns composition and command flow; the reference owns public syntax. |
|
||||
| Building a subprocess caller or changing its result protocol | [Subprocess Consumer Guide](consumers/subprocess.md), [Run Result Receipt](integrations/run-result.md), and [CLI Internals](internal/cli.md) | These separate caller workflow, durable receipt contract, and CLI implementation behavior. |
|
||||
| Building a subprocess caller or changing its result protocol | [Subprocess Consumer Guide](consumers/subprocess.md), [Complete D&D Consumer Guide](consumers/dnd-pipeline.md), [Run Result Receipt](integrations/run-result.md), and [CLI Internals](internal/cli.md) | These separate generic caller workflow, the complete D&D workflow, the durable receipt contract, and CLI implementation behavior. |
|
||||
| Configuration loading, resolution, or user-visible configuration behavior | [Configuration Internals](internal/configuration.md) and [Configuration](config.md) | The internal guide owns loading and resolution mechanics; the reference owns the configuration contract. |
|
||||
| Pipeline resolution or execution | [Pipeline Internals](internal/pipeline.md) | It documents profiles, references, validation, retries, checkpoints, and runner behavior. |
|
||||
| Production modules or validators | [Module Internals](internal/modules.md), [D&D Module Internals](internal/dnd.md), and [D&D integration contracts](integrations/) | The generic guide owns extension mechanics, the D&D guide owns shared family conventions, and the contracts own durable output shapes. |
|
||||
|
||||
@@ -1,9 +1,11 @@
|
||||
# Published Evidence Context
|
||||
|
||||
This contract defines the optional `source/evidence-context` artifact emitted
|
||||
by the production JSON output. Its configuration is owned by
|
||||
[Configuration](../config.md#module-bindings-and-validators); its logical-file
|
||||
discovery is owned by [Published JSON Output](json-output.md).
|
||||
by the production JSON output. It is a selected source-unit excerpt for
|
||||
convenient reading alongside normalized lane artifacts; it is not a second
|
||||
citation or provenance model. Its configuration is owned by
|
||||
[Configuration](../config.md#module-bindings-and-validators), and its
|
||||
logical-file discovery is owned by [Published JSON Output](json-output.md).
|
||||
|
||||
## Identity And Discovery
|
||||
|
||||
@@ -26,91 +28,80 @@ its absence means evidence publication was not enabled for that bundle.
|
||||
|
||||
## Payload
|
||||
|
||||
The v1 payload is a JSON object with required `source_id`, `source_digest`,
|
||||
`window_units`, `selected_lanes`, and `contexts` fields. `selected_lanes` and
|
||||
`contexts` are always arrays; an enabled configuration with no accepted direct
|
||||
evidence publishes `contexts: []`.
|
||||
The v1 payload is a top-level JSON array of generic source units. There is no
|
||||
wrapper, source-level metadata, context grouping, lane identifier, or evidence
|
||||
reference in the payload. An enabled configuration with no contributing
|
||||
accepted evidence publishes `[]`.
|
||||
|
||||
```json
|
||||
{
|
||||
"source_id": "session-alpha",
|
||||
"source_digest": "sha256:0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef",
|
||||
"window_units": 1,
|
||||
"selected_lanes": ["npc_registry", "spells"],
|
||||
"contexts": [
|
||||
{
|
||||
"context_ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 10,
|
||||
"end_unit_id": 20
|
||||
},
|
||||
"evidence_refs": [
|
||||
{
|
||||
"lane_id": "spells",
|
||||
"source_ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 10,
|
||||
"end_unit_id": 10
|
||||
}
|
||||
}
|
||||
],
|
||||
"units": [
|
||||
{
|
||||
"id": 10,
|
||||
"kind": "transcript_segment",
|
||||
"text": "Aria casts Cure Wounds.",
|
||||
"ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 10,
|
||||
"end_unit_id": 10
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 20,
|
||||
"kind": "transcript_segment",
|
||||
"text": "The party regroups.",
|
||||
"ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 20,
|
||||
"end_unit_id": 20
|
||||
}
|
||||
}
|
||||
]
|
||||
[
|
||||
{
|
||||
"id": 10,
|
||||
"kind": "transcript_segment",
|
||||
"text": "Aria casts Cure Wounds.",
|
||||
"ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 10,
|
||||
"end_unit_id": 10
|
||||
}
|
||||
]
|
||||
}
|
||||
},
|
||||
{
|
||||
"id": 20,
|
||||
"kind": "transcript_segment",
|
||||
"text": "The party regroups.",
|
||||
"ref": {
|
||||
"source_id": "session-alpha",
|
||||
"start_unit_id": 20,
|
||||
"end_unit_id": 20
|
||||
}
|
||||
}
|
||||
]
|
||||
```
|
||||
|
||||
Each context requires a `context_ref` object and `evidence_refs` and `units`
|
||||
arrays. `context_ref` identifies the first and last included unit. Each
|
||||
evidence entry contains a selected `lane_id` and an original `source_ref`. A
|
||||
unit uses the existing source-unit shape: required `id`, `kind`, `text`, and
|
||||
self `ref`, plus optional JSON-object `metadata`. Fixed payload objects reject
|
||||
unknown fields; unit metadata may contain application-defined JSON values.
|
||||
Each source unit has required `id`, `kind`, `text`, and self `ref` fields.
|
||||
`ref` contains `source_id`, `start_unit_id`, and `end_unit_id`, and both unit
|
||||
endpoints identify that unit's `id`. A unit may also contain source-owned
|
||||
`metadata`, an open-ended JSON object. Fixed unit and reference fields are
|
||||
strict: consumers must reject unknown fixed fields, malformed units, invalid
|
||||
self-references, units whose `source_id` differs from other units in the same
|
||||
excerpt, and a payload that is not the array described here.
|
||||
|
||||
## Citations And Context
|
||||
The excerpt preserves each selected unit exactly as represented by the
|
||||
validated generic source document. It does not add evidence-context-specific
|
||||
annotations or reshape source-owned metadata.
|
||||
|
||||
`evidence_refs` are the authoritative citations. They identify the direct
|
||||
references emitted by accepted normalized artifacts. `context_ref` and the
|
||||
units collection include those cited units plus nearby source units selected by
|
||||
the configured window. They are explanatory context, not widened citations.
|
||||
## Selection And Citations
|
||||
|
||||
Only accepted outputs from the configured lane allowlist contribute. Rejected,
|
||||
failed, absent, and lane-filtered outputs do not contribute. The artifact never
|
||||
contains raw input bytes, prompts, model responses, auxiliary reference
|
||||
content, credentials, or filesystem paths.
|
||||
The framework obtains direct source references only through typed evidence
|
||||
projections of accepted normalized artifacts in the configured lane allowlist.
|
||||
It validates each reference against the current source document, expands its
|
||||
range by `window_units` source-unit positions on each side, clamps at document
|
||||
boundaries, and takes the union of all expanded ranges. The output contains
|
||||
each selected source unit once in source-document position order, regardless
|
||||
of numeric unit IDs. Repeated references, overlapping windows, and citations
|
||||
from multiple lanes do not duplicate a unit. Rejected, failed, absent,
|
||||
inactive, and unselected lanes contribute nothing.
|
||||
|
||||
## Ordering And Compatibility
|
||||
Normalized lane artifacts remain authoritative for citations and for which lane
|
||||
cited a range. The excerpt has no lane attribution and must not be used to
|
||||
reconstruct it. Its included nearby units provide reading context only; they
|
||||
do not widen any citation in a lane artifact.
|
||||
|
||||
The selected lane allowlist is lexical. Contexts and units are in source
|
||||
document position order, not numeric unit-ID order. Direct evidence entries
|
||||
are deterministically ordered by lane and source reference. Overlapping or
|
||||
contiguous windows merge, and each source unit appears at most once in the
|
||||
resulting contexts.
|
||||
The excerpt contains at most every generic source unit once. It can therefore
|
||||
equal the complete generic source document when coverage is broad or the
|
||||
window is large. No byte-, token-, or compression-size guarantee is made, and
|
||||
the framework does not truncate the excerpt to meet an arbitrary size limit.
|
||||
|
||||
## Consumer Responsibilities And Data Handling
|
||||
|
||||
The artifact is additive to the JSON bundle and is not a lane payload,
|
||||
normalized-output count, checkpoint, or generated reference. Consumers that
|
||||
do not need it must tolerate the absent optional descriptor. Consumers that do
|
||||
use it should preserve the artifact and its schema identity with the run
|
||||
provenance, and should treat its source text and metadata as sensitive durable
|
||||
content.
|
||||
do not need it must tolerate an absent descriptor. Consumers that do use it
|
||||
should validate the descriptor and payload before use, retain the artifact with
|
||||
its schema identity when needed for a run record, and read citations from the
|
||||
corresponding normalized lane artifacts.
|
||||
|
||||
The excerpt contains source-unit text and source-owned metadata and is durable
|
||||
output. Treat it as sensitive source content, apply appropriate access controls
|
||||
and retention, and do not assume its selected form is materially smaller or
|
||||
less sensitive than the original input.
|
||||
|
||||
@@ -24,7 +24,7 @@ root for the logical discovery described here.
|
||||
| `warnings.json` | Accepted-output and run warnings. |
|
||||
| `lanes/<safe-lane-id>.json` | One normalized artifact payload for each lane. |
|
||||
| `chunk-map.json` | Optional accepted chunk map, when its export is enabled and available. |
|
||||
| `evidence-context.json` | Optional source-context artifact, when evidence publication is enabled. |
|
||||
| `evidence-context.json` | Optional selected source-unit excerpt, when evidence publication is enabled. |
|
||||
|
||||
JSON files are pretty-printed with a trailing newline. Lane payloads are
|
||||
accepted only when their media type is `application/json`.
|
||||
|
||||
@@ -84,14 +84,15 @@ replace it with a complete profile of the same ID from the configured PromptKit
|
||||
source. Deployment profile selection is documented in
|
||||
[Configuration](../config.md#promptkit-profiles).
|
||||
|
||||
The transcript assets have distinct consumers. Scene chunking consumes the
|
||||
complete-session `common-dnd-transcript-full.md`; extraction prompts consume
|
||||
the current-chunk `common-dnd-transcript-chunk.md`; and NPC, location, and item
|
||||
normalization consume `common-dnd-transcript-windows.md` alongside their
|
||||
candidate collections. Player, party, glossary, and compatible campaign
|
||||
references provide disambiguating context only when declared by the active
|
||||
prompt; they never establish evidence. Reference material is canonically
|
||||
ordered before rendering so equivalent inputs remain stable.
|
||||
The D&D transcript assets have distinct consumers. Scene chunking consumes the
|
||||
complete-session `common-dnd-transcript-full.md`, while extraction prompts
|
||||
consume the current-chunk `common-dnd-transcript-chunk.md`. NPC, location, and
|
||||
item normalization instead mount the generic semantic-reconciliation
|
||||
candidate and transcript-window presentation assets. Player, party, glossary,
|
||||
and compatible campaign references provide disambiguating context only when
|
||||
declared by the active prompt; they never establish evidence. Reference
|
||||
material is canonically ordered before rendering so equivalent inputs remain
|
||||
stable.
|
||||
|
||||
Extraction prompts render the common system and identity messages first, then
|
||||
cached campaign references and the cached chunk transcript. Evidence policy and
|
||||
@@ -101,10 +102,11 @@ reusable extraction prefix identical while preserving the lane-specific suffix.
|
||||
|
||||
Scene chunking intentionally uses a different order: system, cached campaign
|
||||
references, uncached module instructions, then the final ephemeral full
|
||||
transcript. Entity normalization also has its own order: system, uncached
|
||||
module instructions, ephemeral reconciliation policy, uncached candidates, and
|
||||
final ephemeral transcript windows. These orders and cache controls are prompt
|
||||
behavior; change them only through the owning manifest and prompt declaration.
|
||||
transcript. Entity normalization also has its own order: D&D system, mandatory
|
||||
generic protocol, ephemeral domain semantic instructions, generic candidate
|
||||
presentation, and final ephemeral generic transcript windows. These orders and
|
||||
cache controls are prompt behavior; change them only through the owning
|
||||
manifest and prompt declaration.
|
||||
|
||||
## Evidence, Candidates, And Normalization
|
||||
|
||||
@@ -134,12 +136,46 @@ canonicalize display values and evidence, use source-document order for stable
|
||||
output, and issue bounded warnings for changes or collapsed duplicates. NPC,
|
||||
item, and location registry normalizers are intentional exceptions: each first
|
||||
produces a deterministic candidate set, then may use a bounded structured-LLM
|
||||
proposal to reconcile identity groups. The proposal selects supplied
|
||||
descriptors—names with their candidate source references—not durable IDs.
|
||||
Request-local candidate keys may support resolution internally, but are never
|
||||
included in model input or output. Colliding descriptors are ineligible, and
|
||||
invalid or unusable proposals retain the deterministic result with retry or
|
||||
fallback diagnostics; the model does not directly replace durable records.
|
||||
proposal to reconcile identity groups.
|
||||
|
||||
## Semantic Registry Reconciliation
|
||||
|
||||
The three registry normalizers instantiate the domain-neutral
|
||||
`internal/framework/semanticreconcile` engine with default bounds. Each
|
||||
eligible candidate receives a contiguous, one-based `candidate_id` for that
|
||||
request. The model sees that handle, the candidate label and source-free
|
||||
evidence ranges, plus bounded transcript windows; it returns only duplicate
|
||||
groups of supplied handles and one supplied canonical handle per group. It
|
||||
never returns names, evidence, durable IDs, or replacement records. Identical
|
||||
labels and evidence remain independently selectable because their handles are
|
||||
distinct.
|
||||
|
||||
The generic core owns the mandatory handle protocol, candidate and transcript
|
||||
presentation, the private response schema, source-reference validation,
|
||||
candidate and combined-material limits, structured completion, proposal
|
||||
assessment, stable group ordering, and typed plan-application mechanics. The
|
||||
D&D prompt contributes its system message and registry-specific semantic
|
||||
instructions. The generic registrar registers the shared prompt and schema;
|
||||
the D&D registrar registers each consuming prompt and the fallback profile.
|
||||
|
||||
Fewer than two eligible candidates skips the LLM without a semantic warning.
|
||||
An exceeded bound also skips the call and preserves the deterministic
|
||||
preprocessed registry, adding the registry's bounded fallback warning. Invalid
|
||||
structured output or discarded proposal groups use the normalizer's existing
|
||||
retry contract; retry exhaustion preserves the safe deterministic or
|
||||
partially applied result and emits its bounded fallback warning. Provider,
|
||||
transport, cancellation, and context-material failures remain execution
|
||||
errors.
|
||||
|
||||
Application remains typed and registry-owned. All three policies select the
|
||||
canonical member's normalized display name, union member evidence in source
|
||||
order, preserve ungrouped records, and derive durable identity only after
|
||||
consolidation. NPC IDs derive from the final name. Item IDs also derive from
|
||||
the final name, and a typed guard prevents currency aliases from crossing
|
||||
denominations or mixing currency with non-currency records. Location IDs
|
||||
derive from the final name and final evidence, preserving same-name,
|
||||
parent/child, and distinct physical-place identities. Registry warning scopes,
|
||||
reason codes, and postconditions remain outside the generic core.
|
||||
|
||||
## Generated References And Grounding
|
||||
|
||||
|
||||
@@ -142,12 +142,28 @@ arrangement and its data-only boundary are defined by
|
||||
[ADR-0011](../adr/0011-centralize-llm-assets.md), rather than by this runtime
|
||||
guide.
|
||||
|
||||
The generic registrar is the sole production registration owner for the
|
||||
semantic-reconciliation default prompt and private response schema. The
|
||||
domain-neutral reconciliation package also exposes only its mandatory protocol
|
||||
and candidate/transcript presentation files for domain prompt manifests. D&D
|
||||
registry normalizers mount those files while retaining ownership and hashing
|
||||
of their D&D system message, semantic instructions, and complete prompt
|
||||
declaration. The response schema is therefore registered once even though
|
||||
several typed normalizers select it.
|
||||
|
||||
Mounted prompt assets determine a module's fingerprint. The fingerprint hashes
|
||||
only the module and shared files explicitly selected by its manifest, so an
|
||||
unrelated asset does not invalidate a checkpoint. Schema loaders validate JSON,
|
||||
attach identity and digest metadata, make defensive copies, and expose
|
||||
diagnostics without raw schema bytes.
|
||||
|
||||
Semantic-reconciliation normalizers extend this identity with the shared
|
||||
response-schema digest, framework policy version, and complete limit-policy
|
||||
digest. Their manifest metadata records the same content-free prompt, schema,
|
||||
policy, and limit identities together with domain identity and normalization
|
||||
policies. Request-local handles, source material, proposal content, and raw
|
||||
asset bytes are not checkpoint metadata.
|
||||
|
||||
Private response schemas validate a model transport envelope. They are not the
|
||||
durable artifact schema and should not be documented as an external wire
|
||||
contract. Durable formats and compatibility rules remain in the
|
||||
|
||||
@@ -42,14 +42,23 @@ generic source references and must use the codec's exact Go type. It does not
|
||||
interpret surrounding context or publish files; the pipeline validates the
|
||||
capability during preparation and the output boundary owns publication. See
|
||||
the [Published Evidence Context contract](../integrations/evidence-context.md)
|
||||
for the durable result.
|
||||
for the durable source-unit excerpt. Lane artifacts retain citation and lane
|
||||
provenance; the framework does not add either to that published excerpt.
|
||||
|
||||
An artifact family is broader than a module: it owns the cohesive domain
|
||||
feature across its artifact type, codec, stage modules, validators, prompt
|
||||
policy, schemas, identity helpers, and reference projections. An extractor and
|
||||
normalizer in one artifact family remain independently registered modules in
|
||||
their respective pipeline stages. This ownership vocabulary does not create a
|
||||
new registry or change the fixed pipeline.
|
||||
|
||||
## Production Composition
|
||||
|
||||
Production composition is intentionally split by family:
|
||||
|
||||
- The generic registrar provides the unit chunker, generic JSON validators,
|
||||
and JSON output encoder.
|
||||
JSON output encoder, and shared semantic-reconciliation prompt and response
|
||||
schema assets.
|
||||
- The Seriatim registrar provides the transcript input adapter. Its external
|
||||
input behavior is defined by the [Seriatim contract](../integrations/seriatim.md).
|
||||
- The D&D registrar provides its codecs, extractors, mergers, normalizers,
|
||||
@@ -60,6 +69,36 @@ The CLI owns the composition that invokes these registrars. A module package
|
||||
may register its own family but must not assemble the CLI or make framework
|
||||
packages depend on production extensions.
|
||||
|
||||
## Semantic Reconciliation
|
||||
|
||||
`internal/framework/semanticreconcile` is a domain-neutral strategy used by a
|
||||
typed normalize module; it is not itself a selectable stage module. A
|
||||
source-backed artifact-family normalizer projects its deterministic records
|
||||
into contextual candidates and owned typed record envelopes, supplies its
|
||||
chosen prompt identity and resolved LLM profile, and constructs an engine with
|
||||
explicit limits. The core filters invalid evidence, assigns contiguous
|
||||
request-local integer handles, renders bounded candidate and transcript
|
||||
materials, invokes the structured-completion boundary, and assesses the
|
||||
returned duplicate groups into a stable non-overlapping plan.
|
||||
|
||||
The normalizer then applies that plan through a typed `ApplicationPolicy`. The
|
||||
core preserves ungrouped records, contribution order, and provenance while the
|
||||
artifact family owns group guards, field and evidence consolidation, durable
|
||||
ID derivation, retry and fallback presentation, warnings, and postconditions.
|
||||
Request-local handles do not enter the typed value or durable artifact. Fewer
|
||||
than two eligible candidates skips model invocation; exceeding a candidate or
|
||||
combined-material bound preserves the deterministic result under the family's
|
||||
fallback policy. Provider, transport, cancellation, and context-construction
|
||||
failures remain execution errors.
|
||||
|
||||
The core supplies a conservative generic prompt and the single private
|
||||
response schema. A domain prompt may substitute its semantic instructions but
|
||||
mounts the core-owned protocol and candidate/transcript presentation assets.
|
||||
Prompt, schema, policy, and limit identities participate in manifest metadata
|
||||
and checkpoint fingerprints. The generic registrar owns production
|
||||
registration of those shared assets; a consuming domain registrar owns only
|
||||
its domain prompt.
|
||||
|
||||
## Adding Or Changing A Module
|
||||
|
||||
1. Choose the pipeline stage and the typed artifact boundary. Put external
|
||||
|
||||
@@ -29,6 +29,7 @@ physical state roots.
|
||||
| Generic models | **internal/core/source**, **internal/core/artifacts**, **internal/framework/contracts** | Source documents and chunks, manifests and provenance, plus typed artifact, reference, validation, output, and structured-completion contracts. |
|
||||
| Pipeline framework | **internal/framework/pipeline** | Registries, profile and reference resolution, typed preparation, validation, retry coordination, ordered execution, handoff, and result assembly. |
|
||||
| LLM and prompt runtime | **internal/framework/llm**, **internal/framework/promptfs** | Provider-neutral structured completions, scheduling, profile recording, prompt assets, schema registration, and credential-shaped-value redaction. |
|
||||
| Semantic reconciliation | **internal/framework/semanticreconcile** | Bounded source-backed candidate preparation, request-local handle proposals, deterministic assessment, typed plan application, and reconciliation identity metadata; see [Module Internals](modules.md#semantic-reconciliation) and [D&D Module Internals](dnd.md#semantic-registry-reconciliation). |
|
||||
| Embedded LLM content | **assets** | Read-only centralized LLM-facing content, scoped by its consuming package; see [LLM Runtime](llm.md#prompt-and-schema-assets) and [D&D Module Internals](dnd.md#prompt-construction). |
|
||||
| Runtime state | **internal/core/fileio**, **internal/core/debugbundle**, **internal/framework/checkpoint**, **internal/framework/chunkplan**, **internal/framework/chunkmap**, **internal/framework/debug** | Confined atomic files, debug bundles, checkpoint and chunk-plan state, accepted chunk maps, and pipeline-facing debug recording. |
|
||||
| Production extensions | **internal/modules/generic**, **internal/modules/seriatim**, **internal/modules/dnd** | Domain-neutral extensions, Seriatim input support, and D&D extraction families registered into the production catalog. |
|
||||
@@ -49,8 +50,9 @@ the CLI composition boundary.
|
||||
composition, and path safety.
|
||||
- [LLM Runtime](llm.md): structured completion, scheduling, prompt assets,
|
||||
profiles, and secret handling.
|
||||
- [Module Internals](modules.md): generic extension registration, module
|
||||
construction, validation, and reference mechanics.
|
||||
- [Module Internals](modules.md): generic extension registration, artifact
|
||||
families, module construction, semantic reconciliation, validation, and
|
||||
reference mechanics.
|
||||
- [D&D Module Internals](dnd.md): shared D&D extractor conventions, generated
|
||||
reference projections, and lane-specific exceptions. Durable D&D and
|
||||
Seriatim data shapes remain in the [integration contracts](../integrations/).
|
||||
|
||||
@@ -110,9 +110,9 @@ are defined in [Accepted Chunk Map](integrations/chunk-map.md). An optional
|
||||
[evidence context](integrations/evidence-context.md) contains source-unit text
|
||||
and metadata. It is not a cache or debug artifact: retain it with the output
|
||||
bundle only for as long as consumers need it, and apply source-content access
|
||||
controls to the entire bundle. Selected lanes may collectively cite most of a
|
||||
transcript, so a broad allowlist can make the evidence artifact nearly as
|
||||
sensitive and large as the source itself.
|
||||
controls to the entire bundle. Its selected source-unit excerpt may include
|
||||
every source unit once when coverage is broad or its configured window is
|
||||
large, so do not assume a byte or token reduction or reduced sensitivity.
|
||||
|
||||
## Chunk-Plan Cache
|
||||
|
||||
|
||||
@@ -24,6 +24,12 @@ DAGs or a general workflow language. Every stage remains explicit; general
|
||||
chunking, merging, or normalization behavior must not be hidden inside an
|
||||
extractor.
|
||||
|
||||
A stage module is one configured implementation of one pipeline stage. An
|
||||
artifact family is the cohesive domain feature that owns an artifact across
|
||||
the explicit stages and supporting codecs, validators, prompts, identity
|
||||
rules, and reference projections. Artifact-family ownership does not combine
|
||||
stages or alter the fixed pipeline.
|
||||
|
||||
Input and chunking are pipeline-wide. Each selected artifact lane owns its
|
||||
extract, merge, and normalize stages, and the output stage aggregates the run's
|
||||
lane outcomes.
|
||||
@@ -39,6 +45,12 @@ implementations. Domain-neutral model and framework layers provide reusable
|
||||
policy, contracts, and orchestration. Concrete input, pipeline, output, and
|
||||
validation extensions depend inward on those generic layers.
|
||||
|
||||
Semantic reconciliation is one such domain-neutral framework mechanism. It
|
||||
prepares bounded source context, invokes a shared model-judgment protocol,
|
||||
validates proposals, and applies safe plans through typed policies supplied by
|
||||
the consuming artifact family. It does not own domain identity, durable IDs,
|
||||
warning semantics, or artifact construction rules.
|
||||
|
||||
Generic layers must not depend on production extensions. Concrete extensions
|
||||
must not compose the application or take ownership of process behavior. The
|
||||
current packages implementing these layers are inventoried in
|
||||
@@ -186,8 +198,13 @@ or domain-specific prompt logic.
|
||||
When a model selects an application entity, callers must supply a contextual
|
||||
selection and deterministically attach the opaque application identity whenever
|
||||
the selection resolves exactly. Models do not receive or reproduce opaque
|
||||
application identifiers; [ADR-0012](../adr/0012-resolve-opaque-entity-identifiers-deterministically.md)
|
||||
records the rationale and limited request-local-label exception.
|
||||
application identifiers. Semantic reconciliation may instead expose
|
||||
contiguous, one-based candidate handles that exist only for one request;
|
||||
deterministic code resolves them before typed application, and they never
|
||||
become durable identity. This is the approved request-local-label application
|
||||
of [ADR-0012](../adr/0012-resolve-opaque-entity-identifiers-deterministically.md)
|
||||
recorded by
|
||||
[ADR-0013](../adr/0013-use-request-local-candidate-handles-for-semantic-reconciliation.md).
|
||||
|
||||
LLM calls and other external operations accept cancellation and respect
|
||||
timeouts. Concurrency control belongs in shared runtime plumbing rather than in
|
||||
|
||||
@@ -1,516 +0,0 @@
|
||||
# Codebase Audit Plan
|
||||
|
||||
## Purpose
|
||||
|
||||
This document defines a repository-wide audit of Notarius for correctness,
|
||||
efficiency, maintainability, and clarity. The audit should identify concrete
|
||||
improvements without treating abstraction, fewer lines, or higher test coverage
|
||||
as goals in themselves.
|
||||
|
||||
The audit is intentionally separate from implementation. Its findings should
|
||||
be evidence-backed and sufficiently specific to support a later remediation
|
||||
roadmap, but the audit should not modify production code, tests, assets, or
|
||||
current-behavior documentation.
|
||||
|
||||
## Governing Principles
|
||||
|
||||
The audit must preserve the architecture and testing policies in
|
||||
`docs/policy/architecture.md` and `docs/policy/testing.md`.
|
||||
|
||||
In particular:
|
||||
|
||||
- Notarius remains a fixed, staged pipeline rather than a general workflow
|
||||
engine.
|
||||
- Generic framework packages must remain domain-neutral, and production
|
||||
modules must not acquire CLI or physical-state responsibilities.
|
||||
- Typed artifact boundaries, exact codec compatibility, deterministic ordering,
|
||||
whole-output validation, and generated-reference provenance are correctness
|
||||
properties, not incidental complexity to be optimized away.
|
||||
- The root `assets` package remains a content-only dependency leaf.
|
||||
- Shared helpers should protect demonstrated common semantics. Similar-looking
|
||||
code with different ownership, error policy, identity rules, or type contracts
|
||||
should remain separate.
|
||||
- Tests should protect durable behavior and meaningful risks. The audit should
|
||||
not recommend tests merely to increase coverage or freeze implementation
|
||||
details.
|
||||
- Efficiency claims must distinguish measured or structurally credible costs
|
||||
from cosmetic line-count reductions. Optimizing local CPU work that is
|
||||
insignificant beside an LLM call is low priority unless it also simplifies
|
||||
correctness or applies to large inputs.
|
||||
|
||||
## Audit Questions
|
||||
|
||||
Every audited area should be examined through the following questions.
|
||||
|
||||
### Correctness
|
||||
|
||||
- Are documented architecture invariants enforced at the correct boundary?
|
||||
- Can invalid configuration, incompatible artifact types, malformed references,
|
||||
or unavailable dependencies reach execution when they could be rejected
|
||||
during resolution or preparation?
|
||||
- Are nil, empty, absent, rejected, failed, and canceled states distinguished
|
||||
consistently?
|
||||
- Are stored or returned slices, maps, byte slices, options, metadata, source
|
||||
documents, references, and artifacts defensively owned where required?
|
||||
- Are public ordering, selected errors, warnings, and checkpoint decisions
|
||||
deterministic regardless of map or goroutine completion order?
|
||||
- Do cancellation, retry, validation, and partial-work semantics match their
|
||||
documented ownership?
|
||||
- Do checkpoint and chunk-plan identities include every semantic dependency and
|
||||
exclude scheduling-only or diagnostic state?
|
||||
- Can auxiliary references accidentally become source evidence, or can
|
||||
generated references bypass codec, schema, provenance, or step-order checks?
|
||||
- Can provider-specific values, credentials, or source content escape through
|
||||
errors, manifests, debug summaries, cache state, or logs?
|
||||
- Do schemas, codecs, candidate decoders, normalizers, and validators agree on
|
||||
the exact durable contract without silently accepting incompatible shapes?
|
||||
|
||||
### Duplication And Shared Mechanics
|
||||
|
||||
- Which exact or near-duplicate implementations express the same invariant and
|
||||
failure policy?
|
||||
- Has duplicated code already drifted in naming, nil handling, canonicalization,
|
||||
metadata, fingerprints, validation, or diagnostics?
|
||||
- Would a helper have a natural owner and a smaller, clearer contract than the
|
||||
duplicated callers?
|
||||
- Can an extraction preserve static typing and package ownership, or would it
|
||||
require reflection, `any`, callbacks with many policy parameters, or a
|
||||
domain-neutral package importing domain concepts?
|
||||
- Is repeated code required by a small interface adapter or typed registration
|
||||
boundary and therefore clearer when left explicit?
|
||||
|
||||
As a default heuristic, prioritize a shared helper when identical semantics
|
||||
appear in three or more production sites, or in two sites where divergence
|
||||
would create a meaningful correctness risk. Do not use that heuristic as a
|
||||
quota: one substantial duplicate may warrant extraction, while widespread
|
||||
one-line interface methods may not.
|
||||
|
||||
### Simplicity And Idiomatic Go
|
||||
|
||||
- Does a function combine orchestration, policy, transformation, persistence,
|
||||
and reporting that could be separated along existing ownership boundaries?
|
||||
- Are repeated scans, sorts, conversions, clones, encodes, or decodes doing work
|
||||
that can safely occur once?
|
||||
- Are intermediate representations necessary, or can a value be validated,
|
||||
canonicalized, and mapped in one comprehensible pass?
|
||||
- Are maps, sets, stable sorts, generics, standard-library helpers, and error
|
||||
wrapping used idiomatically?
|
||||
- Are abstractions earning their complexity, or are interfaces, option layers,
|
||||
wrappers, aliases, compatibility paths, and private types left over after a
|
||||
completed migration?
|
||||
- Are there unreachable error branches, redundant fingerprints or digests,
|
||||
duplicated sources of truth, or accessors used only by tests?
|
||||
- Can a smaller implementation preserve exact observable behavior and safety
|
||||
properties?
|
||||
|
||||
### Explanatory Comments
|
||||
|
||||
Comments should be recommended where the code is necessarily complex because
|
||||
it preserves a non-obvious invariant. Good candidates include:
|
||||
|
||||
- concurrency coordination, cancellation, and stable error selection;
|
||||
- checkpoint identity, reuse, forced recomputation, and dependency invalidation;
|
||||
- typed erasure and restoration at framework boundaries;
|
||||
- generated-reference ordering and provenance;
|
||||
- canonicalization and identity resolution where registry evidence differs
|
||||
from occurrence evidence;
|
||||
- prompt ordering or input identity required for backend caching; and
|
||||
- path confinement, atomic publication, redaction, or terminal error precedence.
|
||||
|
||||
Recommend comments that explain *why* a step or ordering constraint exists and
|
||||
what would break if it changed. Do not recommend comments that narrate syntax,
|
||||
repeat a function name, duplicate current-behavior documentation, or preserve
|
||||
implementation history.
|
||||
|
||||
### Tests
|
||||
|
||||
- Is each consequential invariant protected at the narrowest stable boundary?
|
||||
- Are concurrency, cancellation, retries, recovery, compatibility, path safety,
|
||||
and data-integrity behavior credibly exercised?
|
||||
- Do higher-level contract tests duplicate lower-level cases without adding
|
||||
integration confidence?
|
||||
- Are tests coupled to private constants, helper shape, exact prose, full error
|
||||
strings, or collaborator choreography rather than behavior?
|
||||
- Can repetitive fixtures or fakes be simplified without creating a test
|
||||
framework more complex than the tests?
|
||||
- Would a focused fuzz test, race test, or package-level invariant test protect
|
||||
a realistic risk better than several example tests?
|
||||
|
||||
## Evidence And Finding Standards
|
||||
|
||||
Static metrics and textual similarity are discovery aids, not findings. A long
|
||||
function may be a clear linear coordinator; identical methods may be useful
|
||||
typed adapters. Every reported finding must include:
|
||||
|
||||
1. a concise title and severity;
|
||||
2. exact files and symbols;
|
||||
3. the observed behavior or structural evidence;
|
||||
4. the correctness, efficiency, maintenance, or comprehension impact;
|
||||
5. a concrete recommended direction;
|
||||
6. important invariants the remediation must preserve;
|
||||
7. focused validation that would demonstrate success; and
|
||||
8. whether the recommendation is independent or should be grouped with another
|
||||
finding.
|
||||
|
||||
Use these severities:
|
||||
|
||||
- **High:** a credible risk of corrupt output, unsafe state handling, secret
|
||||
exposure, stale reuse, deadlock, nondeterminism, or violated external
|
||||
contract.
|
||||
- **Medium:** a plausible behavioral defect, meaningful wasted work on common
|
||||
paths, or complexity/duplication likely to cause future correctness drift.
|
||||
- **Low:** a contained simplification, small efficiency improvement, dead code,
|
||||
naming issue, or missing explanation with no current behavioral failure.
|
||||
|
||||
The audit should explicitly record examined areas with no findings. This makes
|
||||
coverage visible and prevents later agents from repeatedly rediscovering the
|
||||
same safe design.
|
||||
|
||||
## Repository Areas
|
||||
|
||||
### 1. Architecture And Dependency Boundaries
|
||||
|
||||
Inspect `docs/policy/architecture.md`, `docs/adr/`, `docs/internal/overview.md`,
|
||||
package imports, module registrars, and the CLI composition root.
|
||||
|
||||
Look for:
|
||||
|
||||
- framework or core code depending on production modules;
|
||||
- modules depending on CLI, physical roots, or provider-specific types;
|
||||
- domain knowledge placed in generic helpers;
|
||||
- duplicated registries or composition policy outside the owning registrar;
|
||||
- abstractions that turn the fixed pipeline into an implicit general graph; and
|
||||
- current code that no longer matches an accepted ADR or documented invariant.
|
||||
|
||||
Graph-reported cross-layer calls must be traced before being classified because
|
||||
tests and interface implementations can resemble dependency inversions without
|
||||
creating a production import violation.
|
||||
|
||||
### 2. Configuration And CLI Composition
|
||||
|
||||
Inspect `internal/core/config`, `internal/cli`, configuration parsing and
|
||||
redaction tests, profile construction, session derivation, catalog assembly,
|
||||
reference overrides, run-result handling, terminal reporting, and maintained
|
||||
example contract tests.
|
||||
|
||||
Pay particular attention to the currently dense paths around
|
||||
`runPipelineCommand`, configuration profile validation, selected reference
|
||||
targets, recomputation policy, and option normalization. Determine whether
|
||||
their complexity reflects necessary composition or mixed responsibilities that
|
||||
can be separated without moving policy into the framework.
|
||||
|
||||
Verify:
|
||||
|
||||
- file, environment, CLI, pipeline, binding, and prompt-default precedence;
|
||||
- consistent strict option and unknown-field handling;
|
||||
- session identity independence from references and pipeline-local changes;
|
||||
- effective profile and runtime fingerprint consistency;
|
||||
- redaction before errors or debug/manifest boundaries;
|
||||
- output publication only after framework success; and
|
||||
- one guarded terminalization path that preserves the primary failure.
|
||||
|
||||
### 3. Pipeline Resolution, Preparation, And Typed Registries
|
||||
|
||||
Inspect `internal/framework/pipeline/profile.go`, `prepare.go`, registry files,
|
||||
`options.go`, `references.go`, `handoff.go`, `construction.go`, typed contracts,
|
||||
and their focused tests.
|
||||
|
||||
This area deserves a dedicated pass because the current graph identifies
|
||||
`ResolvePipeline`, generated-binding validation, reference-target resolution,
|
||||
and generated-reference construction as high-complexity or high-fan-in code.
|
||||
|
||||
Verify:
|
||||
|
||||
- static failures occur before source parsing;
|
||||
- selected and unselected lanes do not contaminate each other's requirements;
|
||||
- stage defaults and overrides have one canonical resolution path;
|
||||
- typed registration and private erasure cannot panic or accept near-matching
|
||||
artifact types;
|
||||
- generated bindings reject cycles, forward references, ambiguity, wrong kinds,
|
||||
and missing accepted normalized producers;
|
||||
- materialized reference bytes and options are cloned and bounded; and
|
||||
- resolved composition and prepared fingerprints include the complete semantic
|
||||
policy exactly once.
|
||||
|
||||
Compare input, chunker, extractor, merger, normalizer, output, validator, codec,
|
||||
evidence-projector, and validator-chain registries for shared mechanics and
|
||||
intentional differences. Repeated typed registration code is a candidate only
|
||||
if a helper can retain useful compile-time guarantees and stage-specific
|
||||
diagnostics.
|
||||
|
||||
### 4. Pipeline Execution, Validation, Retry, And Concurrency
|
||||
|
||||
Inspect `runner*.go`, `typed_execution.go`, `runner_typed.go`,
|
||||
`runner_concurrent.go`, validation-chain execution, normalize retry behavior,
|
||||
synchronized collaborators, and the concurrency, cancellation, retry, debug,
|
||||
and checkpoint tests.
|
||||
|
||||
Trace complete paths rather than reviewing helper files in isolation:
|
||||
|
||||
- source and chunk-plan selection through chunk validation;
|
||||
- deterministic chunk-first/lane-second dispatch;
|
||||
- lane extraction through merge and normalize continuations;
|
||||
- rejection versus framework-error propagation;
|
||||
- cancellation before dispatch, while queued, and while running;
|
||||
- retry attempts and warning retention;
|
||||
- stable error selection after concurrent completion;
|
||||
- checkpoint hydration back into typed execution; and
|
||||
- output suppression after a framework error.
|
||||
|
||||
Look for goroutine leaks, unbounded work, lock-order risks, double release or
|
||||
double recording, races on shared result state, unnecessary serialization,
|
||||
and repeated canonicalization. Comments are especially valuable here when they
|
||||
explain ordering or cancellation invariants that are not apparent from local
|
||||
control flow.
|
||||
|
||||
### 5. State, Checkpoints, Chunk Plans, Debugging, And File Safety
|
||||
|
||||
Inspect `internal/framework/checkpoint`, `chunkplan`, `chunkmap`, `debug`,
|
||||
`evidencecontext`, `internal/core/fileio`, `debugbundle`, and their CLI
|
||||
composition.
|
||||
|
||||
Verify:
|
||||
|
||||
- narrow path validation and symlink-resistant confinement;
|
||||
- atomic writes and recoverable explicit cleanup;
|
||||
- separation of checkpoint recording, resume loading, chunk-plan caching, and
|
||||
debug capture;
|
||||
- canonical encoding before content identity is trusted;
|
||||
- complete but non-secret checkpoint fingerprints;
|
||||
- correct ordinary-resume and selective-recompute behavior;
|
||||
- producer dependency invalidation across ordered steps;
|
||||
- immutable hydration and no aliasing with stored bytes;
|
||||
- debug data never influencing execution or reuse; and
|
||||
- terminal persistence failures never obscuring the primary error.
|
||||
|
||||
Review the repeated extract/merge/normalize recorder and loader methods, path
|
||||
component validators in multiple state packages, and clone/encode/decode paths.
|
||||
Determine which repetition is a clear stage adapter and which can share a
|
||||
private primitive without weakening reason-code ownership or diagnostics.
|
||||
|
||||
### 6. LLM Runtime, Prompt Filesystems, And Assets
|
||||
|
||||
Inspect `internal/framework/llm`, `promptfs`, the PromptKit integration,
|
||||
scheduler, profile-source construction, prompt/schema registries, root
|
||||
`assets`, module prompt manifests, and relevant D&D shared assets.
|
||||
|
||||
Verify:
|
||||
|
||||
- every provider call passes through the shared scheduler and cancellation
|
||||
removes queued calls safely;
|
||||
- PromptKit and Notarius concurrency limits compose as documented;
|
||||
- profile inspection and runtime use identical source precedence;
|
||||
- session IDs, profile-source fingerprints, prompt fingerprints, and schema
|
||||
fingerprints reflect the intended semantic inputs;
|
||||
- secrets and provider-specific error types do not cross the boundary;
|
||||
- prompt inputs and private outputs do not expose opaque entity IDs;
|
||||
- prompt ordering, stable prefixes, and cache controls remain intentional;
|
||||
- schema loaders and filesystem adapters validate once and return defensive
|
||||
data; and
|
||||
- the root assets package contains no business logic.
|
||||
|
||||
Compare the LLM asset registry and prompt-filesystem adapters for duplicated
|
||||
filesystem behavior. Review repeated prompt/schema loader and metadata code in
|
||||
module packages, but reject an extraction that would centralize domain prompt
|
||||
ownership or make unrelated assets share one invalidation boundary.
|
||||
|
||||
### 7. Generic And Seriatim Modules
|
||||
|
||||
Inspect `internal/modules/generic` and `internal/modules/seriatim`, including
|
||||
module specs, option decoding, chunk planning, input translation, validators,
|
||||
output encoding, evidence-context publication, registration, and tests.
|
||||
|
||||
Verify that:
|
||||
|
||||
- external Seriatim details end at the input boundary;
|
||||
- generic chunking and output remain domain-neutral;
|
||||
- chunk plans and source units preserve source-addressed invariants;
|
||||
- output logical names are safe and deterministic;
|
||||
- output options do not bypass preparation-time compatibility checks; and
|
||||
- option decoding is strict, small, and consistent with configuration
|
||||
validation.
|
||||
|
||||
The graph flags generic integer option parsing and JSON output policy decoding
|
||||
as relatively complex. Examine whether that is inherent strict decoding or an
|
||||
opportunity for a smaller typed parser with equally precise diagnostics.
|
||||
|
||||
### 8. D&D Domain Model, Codecs, And Shared Helpers
|
||||
|
||||
Inspect `internal/modules/dnd` domain types, codecs, candidate decoders,
|
||||
identity packages, registries, shared source-reference helpers, diagnostics,
|
||||
registry resolution, entity reconciliation, mergers, registrar, and assets.
|
||||
|
||||
Compare all ten current artifact families. Build a convention matrix covering:
|
||||
|
||||
- module specs and execution classes;
|
||||
- constructor and option behavior;
|
||||
- manifest metadata and checkpoint fingerprints;
|
||||
- response-schema loading and private-versus-durable types;
|
||||
- source-reference conversion, canonicalization, ordering, and deduplication;
|
||||
- nil versus present-empty output;
|
||||
- codecs and strict JSON behavior;
|
||||
- registry lookup, identity derivation, immutable projections, and resolution;
|
||||
- normalizer retry/fallback behavior;
|
||||
- validators and default chains; and
|
||||
- registration, prompt assets, and documentation ownership.
|
||||
|
||||
The graph reports many exact similarities among codec `Decode` methods,
|
||||
fingerprint/metadata methods, registry extractors, identity helpers, occurrence
|
||||
normalizers, and validators. Treat these as a prioritized review list, not an
|
||||
instruction to create one generic D&D engine. A worthwhile helper must preserve
|
||||
domain-specific identity, evidence, kind ordering, validation, diagnostics,
|
||||
and artifact typing.
|
||||
|
||||
### 9. D&D Extraction And Normalization Flows
|
||||
|
||||
Trace each lane end to end rather than auditing only similarly named files:
|
||||
|
||||
- spells;
|
||||
- NPC registry and NPC occurrences;
|
||||
- combat turns and enemy events;
|
||||
- item registry and item occurrences;
|
||||
- scene descriptions; and
|
||||
- location registry and location occurrences.
|
||||
|
||||
For registry/occurrence pairs, verify the complete semantic boundary: the model
|
||||
uses contextual evidence, deterministic code attaches opaque identity, registry
|
||||
evidence does not become occurrence evidence, and unresolved or ambiguous
|
||||
selections fail according to lane policy.
|
||||
|
||||
Review whether any lane resolves or canonicalizes the same entity, source
|
||||
reference, or response twice; constructs unnecessary intermediate response
|
||||
forms; performs repeated sorts or scans; or retains transitional paths. Compare
|
||||
registry normalizers and occurrence normalizers for genuinely identical
|
||||
mechanics, while keeping currency, same-name location, NPC ambiguity, spell
|
||||
catalog, scene eligibility, and combat-specific policy with their owners.
|
||||
|
||||
### 10. Test Suite And Comment Coverage
|
||||
|
||||
Review the tests associated with every preceding area after understanding the
|
||||
production contracts. This should be a cross-cutting pass, not a request to add
|
||||
tests for every flagged function.
|
||||
|
||||
Identify:
|
||||
|
||||
- consequential unprotected invariants;
|
||||
- duplicated policy assertions across layers;
|
||||
- brittle tests coupled to internal constants, prompt prose, or private helper
|
||||
shape;
|
||||
- oversized test harnesses and repeated fixtures that obscure intent;
|
||||
- race-sensitive code not exercised under `-race`;
|
||||
- parsers, canonicalizers, and path handlers where fuzzing would address a real
|
||||
input-space risk; and
|
||||
- complex production code whose tests reveal an unclear ownership boundary.
|
||||
|
||||
Also identify necessarily complex symbols that lack a concise invariant-level
|
||||
comment. Comment recommendations should name the exact symbol and the fact the
|
||||
comment should explain; “add more comments” is not an actionable finding.
|
||||
|
||||
## Efficiency Evaluation
|
||||
|
||||
The audit should consider both runtime and maintenance efficiency.
|
||||
|
||||
For runtime efficiency, examine algorithmic behavior relative to realistic
|
||||
input dimensions: source units, chunks, lanes, references, artifacts, registry
|
||||
records, checkpoint files, and prompt assets. Prioritize repeated full-input
|
||||
passes, nested linear lookup, unnecessary JSON round trips, repeated hashing,
|
||||
large defensive copies at adjacent ownership boundaries, and serialization on
|
||||
concurrent hot paths. Preserve a defensive copy when it establishes ownership;
|
||||
removing it solely to reduce allocation is not an improvement.
|
||||
|
||||
For maintenance efficiency, prioritize repeated policy, parallel type systems,
|
||||
duplicated error classification, scattered defaults, and migrations that left
|
||||
two ways to perform the same operation. Boilerplate is costly only when it can
|
||||
drift or obscures the semantic core. Small explicit typed adapters can be more
|
||||
maintainable than a generic abstraction.
|
||||
|
||||
Do not recommend caching, pooling, concurrency, or a benchmark without naming
|
||||
the workload and risk it addresses. Add a benchmark only when a proposed
|
||||
optimization concerns a repeatable local path and the result would influence
|
||||
the decision.
|
||||
|
||||
## Audit Method
|
||||
|
||||
Each area should use the same method:
|
||||
|
||||
1. Read its architecture/internal documentation and focused tests.
|
||||
2. Map public/package contracts and trace the main call paths.
|
||||
3. Inspect high-fan-in, high-cognitive-complexity, nested-loop, and repeated-
|
||||
conversion symbols.
|
||||
4. Review exact and near-duplicate code side by side, including callers and
|
||||
failure semantics.
|
||||
5. Check dependency direction, ownership, aliasing, deterministic order,
|
||||
cancellation, and error classification.
|
||||
6. Compare tests with the risks owned at that layer.
|
||||
7. Record findings and inspected-with-no-finding areas before moving on.
|
||||
8. Run focused read-only validation when it can confirm or refute a suspected
|
||||
problem.
|
||||
|
||||
Prefer the repository knowledge graph for symbol discovery, call tracing, and
|
||||
similarity candidates. Use textual search for literals, diagnostics, config
|
||||
keys, asset content, and stale names. Read complete implementations and tests
|
||||
before reporting a metric-derived candidate.
|
||||
|
||||
## Execution Sequence
|
||||
|
||||
This document owns audit scope, questions, evidence standards, and the quality
|
||||
bar. [Staged Codebase Audit Sequence](audit-sequence.md) is the sole canonical
|
||||
owner of prompt order, stage boundaries, per-stage reading, validation commands,
|
||||
and acceptance criteria. Do not derive or maintain a second sequence here.
|
||||
|
||||
The audit is executed as bounded prompts and writes its accumulated findings to
|
||||
`docs/roadmap/audit.md`. Later stages must build on and reconcile earlier
|
||||
evidence rather than concatenate independent reports. Implementation and
|
||||
roadmap retirement remain separate work after maintainers review the completed
|
||||
audit.
|
||||
|
||||
## Baseline And Validation
|
||||
|
||||
Before the first audit stage, record the commit under review and require a clean
|
||||
worktree. Refresh the code knowledge graph so renamed or deleted code does not
|
||||
produce false findings. Run the normal offline baseline:
|
||||
|
||||
```sh
|
||||
go test ./...
|
||||
go vet ./...
|
||||
go build ./cmd/notarius
|
||||
git diff --check
|
||||
```
|
||||
|
||||
Run `go test -race` for packages with concurrency or mutable shared state,
|
||||
especially `internal/framework/pipeline`, `internal/framework/llm`, state
|
||||
packages, and D&D registries. A repository-wide race run is appropriate for
|
||||
final verification if its cost remains reasonable.
|
||||
|
||||
Optional diagnostic commands should be used only when relevant:
|
||||
|
||||
- `go test -count=1` to rule out cache-masked failures;
|
||||
- `go test -shuffle=on` to detect order coupling;
|
||||
- focused fuzzing for existing or newly justified fuzz targets; and
|
||||
- focused benchmarks or profiles for a specific efficiency finding.
|
||||
|
||||
The audit itself should not change tests to make the baseline pass. Record any
|
||||
pre-existing failure and distinguish it from an audit finding.
|
||||
|
||||
## Deliverable Quality Bar
|
||||
|
||||
The completed audit should:
|
||||
|
||||
- cover every repository area listed above;
|
||||
- distinguish defects from refactoring opportunities and comment requests;
|
||||
- distinguish credible performance costs from aesthetic simplification;
|
||||
- identify intentional duplication that should remain explicit;
|
||||
- avoid recommendations that violate dependency direction or weaken typing;
|
||||
- cite exact evidence and preserve named invariants for every finding;
|
||||
- consolidate root causes rather than report many symptoms;
|
||||
- rank independent work so a later implementation plan can stage it safely;
|
||||
- recommend no code change whose expected benefit is smaller than its added
|
||||
abstraction or test-maintenance cost; and
|
||||
- leave implementation and roadmap retirement to later work.
|
||||
|
||||
## Open Questions
|
||||
|
||||
None are required to begin the audit. If a later stage cannot determine whether
|
||||
behavior is intentional from code, tests, policies, ADRs, or current
|
||||
documentation, it should record the uncertainty and a recommended resolution
|
||||
rather than silently treating preference as a defect.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,350 +0,0 @@
|
||||
# Contextual Entity Grounding
|
||||
|
||||
## Purpose
|
||||
|
||||
Notarius should use an LLM for semantic interpretation of source evidence, not
|
||||
for referential-integrity work that deterministic code can perform more
|
||||
reliably. D&D prompts must therefore stop requiring models to reproduce opaque
|
||||
machine identifiers such as hash-derived entity IDs. Models should identify
|
||||
entities through human-readable, evidence-grounded context, after which
|
||||
Notarius resolves the selection and attaches the canonical durable identity.
|
||||
|
||||
This roadmap defines the policy, affected D&D prompt families, and intended
|
||||
end state. The ordered work needed to reach that state is maintained in
|
||||
[Implementation Plan](implementation.md).
|
||||
|
||||
## User Intent
|
||||
|
||||
The change has two goals:
|
||||
|
||||
- prevent otherwise useful model responses from failing because a long,
|
||||
non-semantic string was copied incorrectly; and
|
||||
- avoid spending prompt space and model effort on exact-copy work that provides
|
||||
no semantic value.
|
||||
|
||||
The policy is not a ban on identifiers. Durable artifacts may continue to use
|
||||
application-owned IDs, and prompts may continue to request source-unit ranges
|
||||
that locate evidence. The policy governs which identity work is assigned to
|
||||
the model.
|
||||
|
||||
## Policy
|
||||
|
||||
An LLM-facing prompt input or private response schema must not require a model
|
||||
to reproduce an opaque machine identifier when Notarius can establish the same
|
||||
association deterministically.
|
||||
|
||||
Opaque machine identifiers include cryptographic hashes, UUIDs, digests,
|
||||
database keys, durable entity IDs, and other tokens whose characters do not
|
||||
carry source-grounded meaning for the model. These values may remain in
|
||||
application state, provenance, diagnostics, checkpoints, and durable artifact
|
||||
contracts, but should be omitted from model-visible material when they do not
|
||||
help the model make a semantic decision.
|
||||
|
||||
The intended responsibility boundary is:
|
||||
|
||||
- the model decides which contextual entity is supported by the supplied
|
||||
evidence and returns the bounded semantic facts requested by the module;
|
||||
- the calling module validates that the contextual selection resolves to
|
||||
exactly one supplied candidate;
|
||||
- deterministic code supplies the canonical display value and durable entity
|
||||
ID; and
|
||||
- existing validators continue to enforce referential integrity at later
|
||||
artifact boundaries.
|
||||
|
||||
Transcript `start_unit_id` and `end_unit_id` values are permitted. They are
|
||||
contextual source coordinates and form part of the evidence contract rather
|
||||
than arbitrary identity tokens. Prompt IDs, schema IDs, fingerprints, session
|
||||
IDs, and digests may also remain in runtime metadata that the model is not
|
||||
asked to reproduce.
|
||||
|
||||
Short request-local labels are a narrowly permitted fallback only when a
|
||||
contextual selector cannot uniquely represent the available choices without
|
||||
unreasonable prompt cost. Such a label must be compact, scoped to one request,
|
||||
validated against the supplied candidate set, and never reused as a durable
|
||||
identity. Current D&D occurrence and reconciliation prompts should be designed
|
||||
without this exception; adopting it later requires a concrete demonstrated
|
||||
need and documented rationale.
|
||||
|
||||
## Current State
|
||||
|
||||
The initial NPC, item, and location registry extractors already follow the
|
||||
desired pattern: the model returns contextual names and evidence, and Notarius
|
||||
derives durable IDs afterward. Spells, combat turns, and enemy events use
|
||||
contextual actor names rather than requiring hash-derived NPC IDs.
|
||||
|
||||
Two current prompt families diverge from that pattern:
|
||||
|
||||
1. `dnd/npc-occurrences`, `dnd/item-occurrences`, and
|
||||
`dnd/location-occurrences` place durable registry IDs in model-visible
|
||||
projections and require the private LLM response to repeat those IDs.
|
||||
2. NPC-, item-, and location-registry normalization use the shared entity
|
||||
reconciliation prompt, which labels candidates with opaque
|
||||
`candidate-000001`-style keys and requires the model to copy those keys into
|
||||
duplicate-group proposals.
|
||||
|
||||
The durable occurrence artifacts correctly retain canonical entity IDs. The
|
||||
problem is the private model transport contract, not the published artifact
|
||||
contract.
|
||||
|
||||
## Target Architecture
|
||||
|
||||
### Model proposals and durable artifacts
|
||||
|
||||
Private LLM response types must express contextual semantic proposals rather
|
||||
than reuse the durable artifact type when that type contains an opaque entity
|
||||
ID. The extractor maps a validated private response into the existing durable
|
||||
artifact only after identity resolution succeeds.
|
||||
|
||||
No affected durable artifact kind, media type, schema ID, schema version, or
|
||||
JSON field changes as part of this work. NPC, item, and location occurrence
|
||||
artifacts continue to publish their exact canonical ID/name pair. Registry
|
||||
artifacts likewise retain their IDs and evidence.
|
||||
|
||||
The private schemas and prompt declarations may remain at their current `v1`
|
||||
identities because Notarius is pre-release and these are not external
|
||||
contracts. Their content hashes, mapping-policy fingerprints, and affected
|
||||
prompt fingerprints must change so incompatible checkpoints are not reused.
|
||||
|
||||
### NPC occurrence grounding
|
||||
|
||||
The NPC occurrence prompt receives an ordered names-only projection of the
|
||||
normalized NPC registry. Its private response contains the canonical NPC name,
|
||||
occurrence kind, and current-transcript source ranges, but no `npc_id`.
|
||||
|
||||
The extractor resolves the returned name under the existing NPC comparison
|
||||
policy. Resolution must produce exactly one registry entry. It then writes that
|
||||
entry's canonical display name and durable ID into the `dnd.NPCOccurrence`.
|
||||
An unknown or ambiguous selection invalidates the extraction operation; the
|
||||
extractor must not guess, use fuzzy matching, silently omit the record, or
|
||||
accept a partial response.
|
||||
|
||||
### Item occurrence grounding
|
||||
|
||||
The item occurrence prompt receives an ordered names-only projection of the
|
||||
normalized item registry. Its private response contains the canonical item
|
||||
name, occurrence kind, kind-specific fields, and current-transcript source
|
||||
ranges, but no `item_id`.
|
||||
|
||||
The extractor resolves the returned name under the existing item comparison
|
||||
and identity policies. Resolution must produce exactly one registry entry,
|
||||
whose canonical name and durable ID are attached deterministically. Unknown or
|
||||
ambiguous selections invalidate the complete extraction operation rather than
|
||||
being guessed, repaired by similarity, or dropped.
|
||||
|
||||
### Location occurrence grounding
|
||||
|
||||
Location identity cannot always be resolved from a display name alone: the
|
||||
current registry intentionally permits same-name locations with distinct
|
||||
source anchors. The location occurrence prompt must therefore receive a
|
||||
contextual registry descriptor that contains the canonical display name plus
|
||||
the minimum source-grounded registry evidence needed to distinguish same-name
|
||||
records. It must not contain the durable `location:sha256:...` value.
|
||||
|
||||
The private response uses two required selector fields: `name` and
|
||||
`registry_refs`. For a comparison-unique canonical name, `registry_refs` is an
|
||||
empty array and Notarius resolves the name under the location comparison
|
||||
policy. For a name shared by multiple registry records, `registry_refs`
|
||||
contains that record's complete canonically ordered registry ranges as
|
||||
`start_unit_id` and `end_unit_id` pairs, without `source_id`.
|
||||
|
||||
Every model-facing registry entry uses one fixed shape with required `name`,
|
||||
`registry_refs`, and `context` fields. `context` is an array of strict objects
|
||||
containing only `unit_id` and `text`. Comparison-unique entries use empty
|
||||
`registry_refs` and `context` arrays. Same-name entries use the complete
|
||||
registry-range selector and the bounded context described below. The model
|
||||
returns only `name` and `registry_refs`; it does not reproduce `context`.
|
||||
|
||||
For same-name groups, the projection also supplies bounded transcript units
|
||||
covered by each record's registry ranges so the model receives meaningful
|
||||
identity context rather than coordinates alone. Those ranges must resolve
|
||||
against the current source document, and the resulting contextual selectors
|
||||
must be unique. An invalid range or selector collision prevents the LLM call
|
||||
and fails the operation. Unique-name entries do not repeat registry ranges or
|
||||
context in the selector, preserving compatibility with a valid registry from
|
||||
another source when the name alone is unambiguous.
|
||||
|
||||
The private response separately supplies current-transcript `source_refs` that
|
||||
prove the occurrence. Registry identity evidence and occurrence evidence must
|
||||
remain different fields and must never be merged. The model should omit an
|
||||
occurrence when the transcript does not support choosing among same-name
|
||||
locations. If a returned selector does not resolve to exactly one supplied
|
||||
registry record, the extractor invalidates the complete operation rather than
|
||||
guessing.
|
||||
|
||||
### Registry reconciliation
|
||||
|
||||
The shared entity-reconciliation input replaces opaque candidate keys with
|
||||
contextual candidate descriptors. At minimum, a descriptor contains the
|
||||
candidate's display name and its canonically ordered source-reference ranges;
|
||||
the existing transcript windows remain available for semantic judgment.
|
||||
|
||||
Duplicate-group members and the canonical member in the private response use
|
||||
the same contextual descriptor shape. The shared reconciliation helper maps
|
||||
each descriptor back to exactly one internal candidate before assessing the
|
||||
proposal. Exact deterministic duplicates should already be removed before the
|
||||
LLM call; any remaining descriptor collision makes the affected candidate
|
||||
ineligible for model-assisted reconciliation rather than authorizing an
|
||||
arbitrary choice.
|
||||
|
||||
Existing safety behavior remains in force: groups must contain at least two
|
||||
supplied candidates, the canonical candidate must be a member, groups must not
|
||||
overlap, and domain-specific eligibility rules remain authoritative. Invalid,
|
||||
ambiguous, or unsafe groups are discarded through the existing bounded
|
||||
fallback and diagnostic behavior. The model never directly mutates the
|
||||
durable registry.
|
||||
|
||||
The shared private reconciliation schema and helper must remain domain-neutral
|
||||
within the D&D family. NPC-, item-, and location-specific duplicate policy
|
||||
continues to live in the owning normalizer.
|
||||
|
||||
## Prompt And Asset Changes
|
||||
|
||||
The following LLM-facing assets are in scope:
|
||||
|
||||
- the prompt instructions, registry input fragments, and private response
|
||||
schemas for NPC, item, and location occurrences;
|
||||
- the prompt manifests where input shape or selected fragments change;
|
||||
- the shared D&D entity-reconciliation fragment and private response schema;
|
||||
and
|
||||
- the NPC-, item-, and location-registry normalization prompt inputs that use
|
||||
the shared reconciliation contract.
|
||||
|
||||
Affected projections must exclude durable entity IDs rather than merely stop
|
||||
mentioning them in prose. Prompt instructions should describe the contextual
|
||||
selection rule once at the narrowest owning asset and must preserve the current
|
||||
distinction between registry grounding and transcript evidence.
|
||||
|
||||
Prompt ordering and cache controls should remain unchanged unless the new
|
||||
contextual input requires an intentional manifest change. Unrelated shared
|
||||
prompt bytes should not be edited. Prompt and schema fingerprints should
|
||||
invalidate only the operations whose selected assets or mapping semantics
|
||||
changed.
|
||||
|
||||
## Code And Validation Changes
|
||||
|
||||
The occurrence extractors need private response types and deterministic
|
||||
registry-resolution paths appropriate to their domain. Shared code is
|
||||
appropriate only for demonstrated mechanics that have identical semantics;
|
||||
NPC, item, and location ambiguity policies must not be forced behind a generic
|
||||
resolver merely to reduce line count.
|
||||
|
||||
Registry projections should expose explicit model-facing methods whose names
|
||||
describe whether they are names-only or contextual identity projections. The
|
||||
existing ID/name projections may remain only for deterministic consumers that
|
||||
genuinely require them; they must no longer be wired to an LLM input.
|
||||
|
||||
Mapping-policy and normalization-policy identifiers must be reviewed and
|
||||
advanced wherever their semantics change. Checkpoint fingerprints must cover
|
||||
the new projection content, private schema, prompt assets, and mapping policy,
|
||||
while continuing to exclude irrelevant internal implementation details.
|
||||
|
||||
Durable occurrence normalizers and registry validators remain defense in
|
||||
depth. They continue to validate exact ID/name pairs on artifacts entering
|
||||
through checkpoints, codecs, or other boundaries even though the LLM no longer
|
||||
produces the ID directly.
|
||||
|
||||
## Testing And Evaluation
|
||||
|
||||
Tests should protect the behavioral boundary rather than prompt prose or
|
||||
private helper structure. The completed work should demonstrate that:
|
||||
|
||||
- affected model-facing registry projections do not contain durable entity
|
||||
IDs;
|
||||
- private occurrence schemas reject opaque ID fields and accept the intended
|
||||
contextual shape;
|
||||
- valid contextual selections map to the exact canonical durable ID/name pair;
|
||||
- unknown, mismatched, and ambiguous selections fail without fuzzy matching,
|
||||
partial acceptance, or arbitrary reassignment;
|
||||
- same-name locations remain distinguishable through contextual evidence;
|
||||
- reconciliation preserves equal-name candidates, resolves valid contextual
|
||||
groups, and discards ambiguous or unsafe proposals;
|
||||
- registry evidence never becomes occurrence evidence;
|
||||
- durable codec, normalization, and validator behavior remains compatible; and
|
||||
- representative assembled D&D pipelines still prepare and execute with fake
|
||||
structured-LLM responses.
|
||||
|
||||
Do not add repository-wide prompt-prose snapshots, exact-message-count tests,
|
||||
or a change-detector test that merely scans for today's field names. Focused
|
||||
projection, schema, mapping, fallback, and integration tests are the stable
|
||||
owners of these risks. Model-quality evaluation with representative
|
||||
transcripts remains a manual development aid rather than an offline test gate.
|
||||
|
||||
## Documentation And Architectural Record
|
||||
|
||||
This policy is durable and applies to future modules, so it warrants
|
||||
`docs/adr/0012-resolve-opaque-entity-identifiers-deterministically.md`, which
|
||||
records:
|
||||
|
||||
- the semantic-proposal versus referential-integrity boundary;
|
||||
- why durable opaque IDs are excluded from model response contracts;
|
||||
- why contextual evidence coordinates remain permitted;
|
||||
- the narrowly scoped request-local-label exception;
|
||||
- alternatives including durable IDs, names-only matching, and short opaque
|
||||
handles; and
|
||||
- the consequences for private schemas, deterministic resolution, debugging,
|
||||
and ambiguous identities.
|
||||
|
||||
`docs/policy/architecture.md` states the general LLM boundary invariant and
|
||||
links to the ADR. `docs/internal/dnd.md` describes the concrete occurrence
|
||||
projections, contextual reconciliation selectors, resolution and failure
|
||||
behavior, and the continued separation of registry grounding from occurrence
|
||||
evidence. `docs/internal/llm.md` contains only a short clarification that
|
||||
caller-owned modules, not PromptKit or the transport adapter, resolve
|
||||
contextual model selections into application identities.
|
||||
|
||||
The NPC, item, and location occurrence and registry integration documents must
|
||||
continue to own their durable wire contracts, while removing current claims
|
||||
that the model-facing consumer projection contains `{id,name}` or that the raw
|
||||
LLM response supplies the durable ID. They should instead explain that
|
||||
Notarius resolves contextual model output and publishes the same exact durable
|
||||
ID/name pair. No public schema examples need to remove those IDs.
|
||||
|
||||
The generic LLM-assisted deduplication entry in `docs/roadmap/future.md` must be
|
||||
reconciled with this policy: stable IDs may exist inside deterministic state,
|
||||
but a future model-facing proposal should use contextual selectors or a
|
||||
documented request-local-label exception rather than durable IDs.
|
||||
|
||||
## Compatibility And Operational Effects
|
||||
|
||||
This work intentionally changes private prompt inputs, private structured
|
||||
responses, and mapping semantics. It will invalidate affected checkpoints
|
||||
through existing prompt, schema, projection, and policy fingerprints. No
|
||||
manual checkpoint migration is required.
|
||||
|
||||
Durable D&D artifacts and generated-reference compatibility remain unchanged.
|
||||
Operators do not receive new configuration fields or CLI controls. The feature
|
||||
does not change PromptKit, provider routing, profile selection, retries,
|
||||
concurrency, or public output placement.
|
||||
|
||||
## Non-Goals
|
||||
|
||||
This work does not:
|
||||
|
||||
- remove canonical IDs from durable registries or occurrence artifacts;
|
||||
- change occurrence categories, evidence rules, or registry identity policy;
|
||||
- add fuzzy, probabilistic, or embedding-based entity resolution;
|
||||
- allow registry provenance to substitute for occurrence evidence;
|
||||
- introduce a general entity graph or cross-artifact identity framework;
|
||||
- redesign unrelated D&D prompts or their schemas;
|
||||
- implement the future generic deduplication normalizer; or
|
||||
- add provider-specific prompt behavior.
|
||||
|
||||
## Acceptance Criteria
|
||||
|
||||
The target state is complete when:
|
||||
|
||||
- no maintained D&D prompt requires a model to reproduce a durable opaque
|
||||
entity ID;
|
||||
- current D&D reconciliation prompts no longer require opaque candidate keys;
|
||||
- NPC, item, and location occurrence LLM outputs are resolved
|
||||
deterministically into their unchanged durable artifacts;
|
||||
- same-name location and reconciliation cases remain safe and unambiguous;
|
||||
- invalid contextual selections preserve the existing extraction-failure or
|
||||
normalization-fallback semantics appropriate to their stage;
|
||||
- affected checkpoint identities change without altering public schema
|
||||
versions;
|
||||
- focused and repository-wide tests pass offline;
|
||||
- the ADR, architecture invariant, D&D internal guide, LLM internal guide,
|
||||
relevant integration contracts, and future roadmap accurately describe
|
||||
their canonical portions of the implemented policy; and
|
||||
- no unrelated code, prompt behavior, or public contract changes are included.
|
||||
224
docs/roadmap/dnd-subprocess-documentation.md
Normal file
224
docs/roadmap/dnd-subprocess-documentation.md
Normal file
@@ -0,0 +1,224 @@
|
||||
# D&D Subprocess Consumer Documentation
|
||||
|
||||
## Status
|
||||
|
||||
Completed. The target guide is `docs/consumers/dnd-pipeline.md`.
|
||||
|
||||
## Purpose
|
||||
|
||||
Provide one task-oriented guide for applications that run Notarius as a
|
||||
subprocess to execute the maintained complete D&D pipeline and consume its
|
||||
published artifacts. The initial concrete consumer is Narratio, but the guide
|
||||
must describe the public Notarius workflow rather than depend on Narratio
|
||||
internals.
|
||||
|
||||
The guide should make the safe integration path obvious without duplicating
|
||||
the CLI, input, receipt, output-bundle, or individual artifact contracts that
|
||||
already have canonical documentation.
|
||||
|
||||
## Current State
|
||||
|
||||
The public integration surface is documented accurately but is distributed
|
||||
across several documents:
|
||||
|
||||
- `docs/consumers/subprocess.md` defines the generic subprocess workflow;
|
||||
- `docs/cli.md` owns commands, flags, stream behavior, and exit statuses;
|
||||
- `docs/integrations/seriatim.md` owns the accepted transcript input shape;
|
||||
- `docs/integrations/run-result.md` owns the machine-readable successful-run
|
||||
receipt;
|
||||
- `docs/integrations/json-output.md` owns bundle discovery and logical files;
|
||||
- the D&D integration documents own the individual lane payload contracts;
|
||||
- `examples/dnd-complete.config.yml` is the maintained complete pipeline.
|
||||
|
||||
A consumer can reconstruct the full workflow from those documents, but there
|
||||
is no D&D-focused guide that connects the maintained example to its input,
|
||||
invocation, complete artifact inventory, discovery procedure, and downstream
|
||||
acceptance decisions.
|
||||
|
||||
## Target Documentation Set
|
||||
|
||||
### Create `docs/consumers/dnd-pipeline.md`
|
||||
|
||||
This document should own the end-to-end consumer workflow for the maintained
|
||||
complete D&D configuration. It should be useful to Narratio and to another
|
||||
subprocess orchestrator with the same needs.
|
||||
|
||||
The guide should contain the following sections.
|
||||
|
||||
#### Prerequisites And Deployment Configuration
|
||||
|
||||
- Link to `examples/dnd-complete.config.yml` rather than embedding a second
|
||||
complete configuration.
|
||||
- Explain that a deployment must provide the configured PromptKit profile and
|
||||
campaign reference files.
|
||||
- Recommend absolute paths for a service or orchestrator deployment.
|
||||
- Call out the path-resolution distinction explicitly: YAML reference paths
|
||||
are relative to the Notarius configuration file, while
|
||||
`promptkit.profile_file` is relative to the Notarius process working
|
||||
directory.
|
||||
- Recommend validating the selected configuration and `dnd-session` pipeline
|
||||
before processing sessions.
|
||||
|
||||
#### Transcript Input
|
||||
|
||||
- State that the complete pipeline consumes a Seriatim JSON document.
|
||||
- Link to the canonical Seriatim contract for required fields and validation.
|
||||
- Recommend the caller's final trimmed transcript when the caller maintains
|
||||
transcript tiers. For Narratio, identify the implemented source as
|
||||
`narratio.transcript.final_trimmed`, normally stored at
|
||||
`transcripts/final.trimmed.json`.
|
||||
- Explain that segment IDs must remain stable because D&D source references
|
||||
cite those units.
|
||||
- Explain that Notarius derives its default prompt session from the input
|
||||
module and exact input bytes and that ordinary callers should not supply
|
||||
`--session-id`.
|
||||
|
||||
#### Subprocess Invocation
|
||||
|
||||
- Show one concise invocation using `notarius run dnd-session`, explicit
|
||||
absolute `--config`, `--input`, and `--output-dir` paths, and `--json`.
|
||||
- Direct callers to capture stdout and stderr separately, propagate
|
||||
cancellation, impose an operator-appropriate timeout, and wait for process
|
||||
completion before parsing stdout.
|
||||
- State that only exit status zero permits receipt decoding and link to the CLI
|
||||
contract for the complete exit-status definition.
|
||||
- Recommend retaining stderr and the invocation context for diagnosis without
|
||||
logging secrets or transcript content.
|
||||
|
||||
#### Receipt And Bundle Discovery
|
||||
|
||||
- Require callers to accept only supported run-result schema versions while
|
||||
tolerating unknown fields allowed by that version.
|
||||
- Direct callers to obtain the exact run-specific bundle from the receipt's
|
||||
absolute `output_directory`; they must not scan for the newest run directory
|
||||
or construct a run ID.
|
||||
- Require a confinement check when resolving `index_file` beneath the reported
|
||||
bundle root.
|
||||
- Direct callers to discover lane payloads by `lane_id` in `index.json`, then
|
||||
verify descriptor media type and schema identity before decoding them.
|
||||
- Explain that descriptor paths are untrusted relative paths and require the
|
||||
same confinement discipline.
|
||||
|
||||
#### Complete D&D Artifact Inventory
|
||||
|
||||
Include a compact table for the ten lane IDs selected by the maintained
|
||||
complete configuration:
|
||||
|
||||
- `item-registry`;
|
||||
- `npc-registry`;
|
||||
- `location-registry`;
|
||||
- `scene-descriptions`;
|
||||
- `item-occurrences`;
|
||||
- `spells`;
|
||||
- `combat-turns`;
|
||||
- `npc-occurrences`;
|
||||
- `location-occurrences`;
|
||||
- `enemy-events`.
|
||||
|
||||
For each row, give a one-line purpose and link to the corresponding canonical
|
||||
D&D artifact contract. Do not copy its fields or schema rules into the
|
||||
consumer guide.
|
||||
|
||||
Document the four always-published bundle files—`index.json`, `manifest.json`,
|
||||
`rejected.json`, and `warnings.json`—and the complete example's configured
|
||||
`chunk-map.json` and `evidence-context.json` pipeline-wide artifacts. Link to
|
||||
their canonical contracts and distinguish pipeline-wide artifacts from lane
|
||||
outputs.
|
||||
|
||||
The inventory must say that a file is available only when its corresponding
|
||||
artifact was accepted and published. It must not imply that process success
|
||||
guarantees every configured lane.
|
||||
|
||||
#### Downstream Acceptance And Retention
|
||||
|
||||
- Explain that exit status zero can coexist with rejected outputs, warnings,
|
||||
or absent lane descriptors.
|
||||
- Require the consumer to define its required lane set explicitly. Recommend
|
||||
treating all ten lanes as required when the caller claims to consume the
|
||||
complete D&D workflow, while allowing another consumer to adopt a narrower
|
||||
documented policy.
|
||||
- Recommend retaining the receipt, the complete published bundle, and captured
|
||||
diagnostic streams long enough to support provenance and failure analysis.
|
||||
- Explain that `evidence-context.json` is a reading excerpt; authoritative
|
||||
citations remain in lane payloads.
|
||||
- Treat transcripts, lane artifacts, evidence context, manifests, and logs as
|
||||
sensitive campaign data.
|
||||
|
||||
#### Compatibility Checklist
|
||||
|
||||
End with a concise checklist covering process exit, receipt schema, path
|
||||
confinement, pipeline identity, index decoding, required descriptors,
|
||||
descriptor schema/media compatibility, warnings and rejections, checksums or
|
||||
retention, and secure handling. Compatibility should be based on published
|
||||
receipt and artifact contracts rather than parsing a human version string.
|
||||
|
||||
### Update Existing Navigation
|
||||
|
||||
- Add a short link from `docs/consumers/subprocess.md` to the D&D-specific
|
||||
workflow. Keep generic subprocess policy in the existing document.
|
||||
- Add the guide to the documentation links in `README.md`.
|
||||
- Extend the subprocess-consumer row in `docs/development.md` so maintainers
|
||||
working on the D&D workflow are routed to the new guide and the canonical
|
||||
contracts.
|
||||
|
||||
### Verify Canonical Contract Documents
|
||||
|
||||
Review the linked integration documents and the complete example while writing
|
||||
the guide. Correct an integration document only if repository inspection finds
|
||||
an actual stale contract. Do not move schema definitions, field tables, CLI
|
||||
flags, or configuration semantics into the new guide.
|
||||
|
||||
## Narratio Alignment
|
||||
|
||||
The guide may name Narratio as the motivating consumer and identify its current
|
||||
final-trimmed transcript source. It must not claim that Narratio already has a
|
||||
Notarius adapter or extraction stage. Until that feature is implemented,
|
||||
Narratio-specific architecture, configuration, stage behavior, manifest
|
||||
records, and artifact source IDs belong in Narratio's roadmap.
|
||||
|
||||
Once Narratio implements the integration, its own integration documentation
|
||||
should link to this guide and the durable Notarius contracts instead of
|
||||
repeating them.
|
||||
|
||||
## Validation
|
||||
|
||||
Documentation implementation should include:
|
||||
|
||||
```sh
|
||||
go run ./cmd/notarius config validate \
|
||||
--config examples/dnd-complete.config.yml \
|
||||
--pipeline dnd-session
|
||||
go test ./...
|
||||
```
|
||||
|
||||
Also verify all new and changed relative Markdown links, compare the artifact
|
||||
inventory directly with the maintained complete configuration, and confirm
|
||||
that commands and path semantics match the CLI and configuration references.
|
||||
If the repository still has no automated link checker, record that fact and
|
||||
perform a focused manual link review.
|
||||
|
||||
## Acceptance Criteria
|
||||
|
||||
- A subprocess integrator can follow one D&D-focused guide from a Seriatim
|
||||
transcript through safe discovery of every artifact configured by the
|
||||
complete example.
|
||||
- The guide makes stdout, stderr, exit-status, receipt, and path-confinement
|
||||
responsibilities unambiguous.
|
||||
- The ten configured D&D lanes and both configured pipeline-wide artifacts are
|
||||
listed and linked to their canonical contracts.
|
||||
- The guide distinguishes process success from the caller's required-artifact
|
||||
policy.
|
||||
- The profile-path and reference-path resolution rules are clearly stated.
|
||||
- Existing navigation makes the guide discoverable.
|
||||
- No volatile contract is defined in two places, and no unimplemented Narratio
|
||||
behavior is presented as current.
|
||||
|
||||
## Non-Goals
|
||||
|
||||
- Implementing or documenting Narratio's future adapter or stage as current
|
||||
Notarius behavior.
|
||||
- Adding a new Notarius command, receipt version, output format, or artifact
|
||||
schema.
|
||||
- Duplicating the complete configuration or individual D&D payload schemas in
|
||||
prose.
|
||||
- Defining a universal partial-result policy for every Notarius consumer.
|
||||
@@ -24,28 +24,52 @@ not as committed release dates.
|
||||
|
||||
## Shared Normalization And Quality Work
|
||||
|
||||
### Generic LLM-Assisted Deduplication
|
||||
The implemented source-backed core and initial D&D registry adoption are
|
||||
described by [Module Internals](../internal/modules.md#semantic-reconciliation)
|
||||
and
|
||||
[D&D Module Internals](../internal/dnd.md#semantic-registry-reconciliation).
|
||||
The [Semantic Reconciliation Roadmap](semantic-reconciliation.md) retains the
|
||||
original feature scope; the sections below keep broader extensions deferred.
|
||||
|
||||
- Add a reusable normalizer that asks an LLM to identify duplicate sets in a
|
||||
list and propose one replacement element for each set.
|
||||
- Define the minimum domain-neutral input contract, initially an ordered list
|
||||
whose elements retain stable unique IDs as internal deterministic state.
|
||||
Model proposals use contextual descriptors, or a specifically justified
|
||||
request-local short label, rather than durable IDs. Artifact-kind
|
||||
registrations or adapters may expose that structure without moving domain
|
||||
rules into the generic package.
|
||||
- Keep mutation deterministic: parse and validate the model's duplicate groups,
|
||||
resolve every supplied descriptor or local label exactly, reject overlapping
|
||||
or malformed groups, prevent unrelated insertion or deletion, and apply only
|
||||
approved replacement operations in code.
|
||||
- Preserve provenance needed for audit and downstream validation, and emit
|
||||
warnings describing every collapsed group.
|
||||
- Evaluate batching and context-window limits before applying the normalizer to
|
||||
large artifact collections.
|
||||
### Large-Collection Semantic Reconciliation
|
||||
|
||||
The model may use its own domain knowledge to judge semantic duplication; the
|
||||
generic implementation is responsible only for the common proposal contract,
|
||||
safety checks, and deterministic application of accepted changes.
|
||||
- Evaluate deterministic candidate blocking only after representative registry
|
||||
inputs exceed the active roadmap's bounded single-request limits. Blocking
|
||||
should use cheap, explainable signals to form plausible comparison sets while
|
||||
preserving the possibility that a duplicate appears outside a lexical name
|
||||
match.
|
||||
- Define correctness for candidates that appear in more than one block,
|
||||
conflicting canonical selections, transitive identity across blocks, retry
|
||||
isolation, and deterministic final ordering before implementation.
|
||||
- Prefer a reconciliation graph or union plan with explicit conflict checks
|
||||
over arbitrary fixed-size slices. Never silently treat a batch boundary as
|
||||
evidence that two candidates are distinct.
|
||||
- Record per-request bounds, block provenance, model calls, discarded
|
||||
proposals, and final group derivation well enough to audit a collapse.
|
||||
|
||||
### Operator-Selected Semantic Policies
|
||||
|
||||
- Consider allowing an operator to select an approved semantic-policy prompt
|
||||
for a typed reconciliation module without replacing the shared protocol,
|
||||
response schema, or deterministic safety rules.
|
||||
- Define the trusted asset source, configuration syntax, compatibility checks,
|
||||
startup validation, provenance, prompt fingerprinting, checkpoint effects,
|
||||
and support boundary before exposing the option.
|
||||
- Prefer selection among registered, typed-policy-compatible prompt assets over
|
||||
arbitrary filesystem prompt paths. Do not add this flexibility until an
|
||||
operator workflow requires it; artifact-family-owned policy remains simpler
|
||||
and safer for the initial implementation.
|
||||
|
||||
### Broader Reconciliation Inputs And Module Selection
|
||||
|
||||
- Revisit alternate context providers when a concrete non-source-backed entity
|
||||
collection needs semantic reconciliation. Any extension must preserve the
|
||||
same request-local identity, deterministic proposal validation, provenance,
|
||||
and typed application guarantees.
|
||||
- Consider a selectable generic normalizer only if Notarius gains a real
|
||||
domain-neutral typed artifact contract that can safely support it. Do not
|
||||
weaken exact artifact registration or introduce reflection-based arbitrary
|
||||
JSON mutation merely to expose a universal module key.
|
||||
|
||||
### Validation And Review
|
||||
|
||||
@@ -102,6 +126,18 @@ checkpoint reuse, when an older artifact may be decoded or adapted, and when a
|
||||
producer or all dependents must be recomputed. Do not add a general migration
|
||||
framework until an actual contract change requires one.
|
||||
|
||||
### Artifact-family-oriented physical packaging
|
||||
|
||||
[ADR-0004](../adr/0004-package-modules-by-domain.md) currently groups production
|
||||
extensions by domain and then by pipeline stage. After artifact-family
|
||||
ownership terminology is established and more families span extraction,
|
||||
normalization, validation, codecs, references, and assets, reassess whether a
|
||||
feature-first physical layout would improve navigation and reduce scattered
|
||||
changes enough to justify a repository-wide package migration. Any change must
|
||||
address Go dependency cycles, registrar ownership, stable public module keys,
|
||||
and supersession of the affected ADR-0004 decision. Conceptual artifact-family
|
||||
ownership does not by itself require this move.
|
||||
|
||||
## Blue-Sky Platform And Operations
|
||||
|
||||
These ideas are intentionally less specified. Promote one into an earlier
|
||||
|
||||
@@ -1,612 +0,0 @@
|
||||
# Contextual Entity Grounding Implementation Plan
|
||||
|
||||
## Objective
|
||||
|
||||
Implement [Contextual Entity Grounding](contextual-entity-grounding.md) so D&D
|
||||
LLM prompts return evidence-grounded contextual selectors while Notarius owns
|
||||
canonical entity IDs and referential integrity. Preserve every durable D&D
|
||||
artifact contract and remove opaque IDs only from model-visible inputs and
|
||||
private model responses.
|
||||
|
||||
This plan is written for a gpt-5.6-terra coding agent. Implement the stages in
|
||||
numeric order. Each stage is intentionally scoped to one implementation prompt
|
||||
and must leave the repository buildable and its focused tests passing before
|
||||
the next stage begins.
|
||||
|
||||
## Plan-Wide Decisions
|
||||
|
||||
Apply these decisions throughout every stage:
|
||||
|
||||
- Read `docs/development.md`, all files under `docs/policy/`, the feature
|
||||
roadmap, and the focused implementation/tests named by the stage before
|
||||
editing.
|
||||
- Preserve the fixed pipeline, typed artifact boundaries, root `assets`
|
||||
content-only rule, module ownership, PromptKit boundary, and evidence rules.
|
||||
- Do not change the durable NPC-, item-, location-registry, or occurrence Go
|
||||
types, JSON schemas, schema IDs, schema versions, media types, reference-slot
|
||||
contracts, categories, or generated-reference compatibility.
|
||||
- Keep the affected prompt and private response-schema identities at `v1`.
|
||||
They are private pre-release transport contracts; their changed content
|
||||
hashes provide the required compatibility boundary.
|
||||
- Advance semantic policy identifiers exactly as directed in each stage. Do
|
||||
not bump unrelated policy identifiers.
|
||||
- A contextual name match uses the entity family's existing comparison policy,
|
||||
never fuzzy matching. A model selection must resolve to exactly one supplied
|
||||
record before a durable ID is attached.
|
||||
- An invalid NPC, item, or location selection invalidates the complete
|
||||
extraction operation. Do not silently drop one response record, accept a
|
||||
partial artifact, or defer a known mapping failure to a later validator.
|
||||
- Registry provenance remains grounding only. Only the current extraction
|
||||
chunk's `source_refs` become occurrence evidence.
|
||||
- Preserve prompt message order and cache controls unless a stage explicitly
|
||||
directs otherwise. Edit only the selected module or shared assets; do not
|
||||
rewrite unrelated shared prompt bytes.
|
||||
- Preserve internal opaque IDs where deterministic code needs them. The rule
|
||||
applies to material shown to the model or requested from it, not to maps,
|
||||
fingerprints, checkpoints, diagnostics, or durable artifacts.
|
||||
- Follow `docs/policy/testing.md`: test package-level behavior and meaningful
|
||||
failure modes, not prompt prose, exact message counts, private helper
|
||||
structure, or a repository-wide string-scanning change detector. All tests
|
||||
remain deterministic, offline, and credential-free.
|
||||
- Use `apply_patch` for edits, `gofmt` changed Go files, and preserve unrelated
|
||||
worktree changes.
|
||||
|
||||
## Final Private Selector Contracts
|
||||
|
||||
These shapes are implementation requirements, not public artifact schemas.
|
||||
|
||||
### NPC occurrence response
|
||||
|
||||
Each response record contains exactly the required fields `name`, `kind`, and
|
||||
`source_refs`. It does not contain `npc_id`. Notarius resolves `name` through
|
||||
the normalized NPC registry and writes the matched registry record's `ID` and
|
||||
canonical `Name` into the durable occurrence.
|
||||
|
||||
### Item occurrence response
|
||||
|
||||
Each response record contains the existing required `name`, `kind`,
|
||||
`quantity`, `from`, `to`, and `source_refs` fields. It does not contain
|
||||
`item_id`. Retain the current nullable representation and kind-specific
|
||||
semantics. Notarius resolves `name` through the normalized item registry and
|
||||
adds the matched `ID` and canonical `Name`.
|
||||
|
||||
### Location occurrence response
|
||||
|
||||
Each response record contains exactly the required fields `name`,
|
||||
`registry_refs`, `kind`, and `source_refs`. `registry_refs` is always an array
|
||||
of strict objects containing required integer `start_unit_id` and
|
||||
`end_unit_id`; it may be empty.
|
||||
|
||||
- When `name` has one comparison-identity match in the supplied registry,
|
||||
`registry_refs` must be empty and name resolution selects that record.
|
||||
- When multiple registry records share the comparison identity,
|
||||
`registry_refs` must equal one record's complete canonically ordered source
|
||||
ranges with `source_id` removed.
|
||||
- The model-facing registry projection uses the same `name` plus
|
||||
`registry_refs` selector and adds a required `context` array. Unique-name
|
||||
records project empty `registry_refs` and `context` arrays. The private
|
||||
response does not reproduce `context`.
|
||||
- Same-name records receive bounded identity context consisting of the ordered
|
||||
source units covered by their registry ranges. Each context element is a
|
||||
strict object with exactly the required fields `unit_id` (integer) and
|
||||
`text` (string); do not expose the durable location ID, source ID, digest,
|
||||
or a replacement token.
|
||||
- Building same-name grounding validates that every registry range belongs to
|
||||
and resolves against the current source. If two records still produce the
|
||||
same contextual selector, grounding construction fails before the LLM call.
|
||||
- `registry_refs` never flow into the durable occurrence's `source_refs`.
|
||||
|
||||
### Entity-reconciliation response
|
||||
|
||||
The shared response remains an object with required `duplicate_groups`.
|
||||
Every group has required `members` and `canonical`. A member and the canonical
|
||||
selection are strict contextual objects containing:
|
||||
|
||||
```json
|
||||
{
|
||||
"name": "Mira Thorn",
|
||||
"source_refs": [
|
||||
{"start_unit_id": 12, "end_unit_id": 12}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
The candidate prompt input uses the same descriptor and contains no `key`.
|
||||
`source_refs` is required and non-empty for every eligible candidate. The
|
||||
shared helper may retain its existing `candidate-000001`-style keys strictly
|
||||
inside Go state to preserve input-position mapping; those keys must never be
|
||||
serialized into prompt input or accepted in the private response.
|
||||
|
||||
If two candidates produce an identical contextual descriptor, neither is
|
||||
eligible for model-assisted reconciliation because the model cannot identify
|
||||
them independently. Otherwise the helper converts returned descriptors to its
|
||||
internal candidate keys before applying all existing unknown-member,
|
||||
ineligible-member, duplicate-member, canonical-membership, overlap, retry, and
|
||||
fallback rules.
|
||||
|
||||
## Stage 1: Convert NPC Occurrences To Name-Based Resolution
|
||||
|
||||
### Goal
|
||||
|
||||
Remove durable NPC IDs from the NPC-occurrence prompt and private response,
|
||||
then resolve the model's contextual name deterministically without weakening
|
||||
checkpoint identity or downstream validation.
|
||||
|
||||
### Work
|
||||
|
||||
1. Inspect:
|
||||
- `assets/dnd/npc-occurrences/`;
|
||||
- `internal/modules/dnd/extract/npcoccurrences/`;
|
||||
- `internal/modules/dnd/npcs/registry/`;
|
||||
- NPC-occurrence normalizer and validator checkpoint fingerprints; and
|
||||
- their focused tests.
|
||||
2. Change `dnd_npc_occurrences_llm.v1.json` so every occurrence requires only
|
||||
`name`, `kind`, and `source_refs`, continues to reject unknown fields, and
|
||||
no longer declares `npc_id`.
|
||||
3. Revise the NPC-occurrence instructions to require a supplied canonical NPC
|
||||
name and current-chunk evidence, with no instruction to copy or invent an
|
||||
ID. Continue using the existing shared names-only NPC registry fragment and
|
||||
preserve manifest order/cache controls.
|
||||
4. Remove `NPCID` from the private `occurrenceResponse`. After canonicalizing
|
||||
response evidence, resolve every response name with the existing
|
||||
`npcregistry.Registry.Lookup` comparison-key lookup. On the first unknown
|
||||
or non-unique selection, return an extractor-scoped mapping error and no
|
||||
value. For a match, construct the durable occurrence with the registry
|
||||
record's exact `ID` and canonical `Name`.
|
||||
5. Change `mappingPolicy` to
|
||||
`dnd.npc_occurrences.extract_mapping.v3`.
|
||||
6. Stop passing `IdentityPromptInput()` to the LLM; use the existing
|
||||
names-only `PromptInput()`.
|
||||
7. Replace the misleading exported model-input API used only for identity
|
||||
fingerprints: retain the unexported ordered `{id,name}` projection and its
|
||||
digest, expose that value as `IdentityDigest() string`, remove
|
||||
`IdentityPromptInput()`, and update NPC-occurrence extractor, normalizer,
|
||||
invariant-validator, and registry-validator fingerprints to use
|
||||
`IdentityDigest()`. The digest must still distinguish ID/name identity from
|
||||
the names-only prompt projection.
|
||||
8. Rewrite existing focused tests around observable behavior: rendered NPC
|
||||
registry input is names-only; the private schema rejects `npc_id`; valid
|
||||
names acquire the registry ID; comparison-equivalent names canonicalize;
|
||||
unknown names fail the whole extraction; registry identity fingerprints
|
||||
remain distinct and defensive; empty registries accept only empty model
|
||||
results. Remove tests whose only purpose was requiring the model to return
|
||||
exact ID/name pairs.
|
||||
|
||||
### Acceptance Criteria
|
||||
|
||||
- No NPC-occurrence LLM input or private response contains a durable NPC ID.
|
||||
- Durable NPC occurrences still contain the exact registry ID/name pair.
|
||||
- Mapping failures remain extractor failures eligible for the configured
|
||||
pipeline retry behavior.
|
||||
- Deterministic consumers still fingerprint the ordered registry identity,
|
||||
while spells, combat turns, enemy events, and NPC occurrences share the
|
||||
names-only model projection.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go fmt ./internal/modules/dnd/npcs/registry ./internal/modules/dnd/extract/npcoccurrences ./internal/modules/dnd/normalize/npcoccurrences ./internal/modules/dnd/validate/npcoccurrences/...
|
||||
go test ./internal/modules/dnd/npcs/registry ./internal/modules/dnd/extract/npcoccurrences ./internal/modules/dnd/normalize/npcoccurrences ./internal/modules/dnd/validate/npcoccurrences/...
|
||||
```
|
||||
|
||||
This stage is suitable for one gpt-5.6-terra prompt.
|
||||
|
||||
## Stage 2: Convert Item Occurrences To Name-Based Resolution
|
||||
|
||||
### Goal
|
||||
|
||||
Give item occurrences the same contextual-name/deterministic-ID boundary while
|
||||
preserving item-specific nullable fields and kind rules.
|
||||
|
||||
### Work
|
||||
|
||||
1. Inspect `assets/dnd/item-occurrences/`, the item occurrence extractor, the
|
||||
item registry, the item occurrence normalizer and registry validator, and
|
||||
their focused tests.
|
||||
2. Change `dnd_item_occurrences_llm.v1.json` to remove `item_id` from required
|
||||
fields and properties. Preserve required `name`, `kind`, `quantity`, `from`,
|
||||
`to`, and `source_refs`, all current enums/nullability, and strict unknown
|
||||
field rejection.
|
||||
3. Rewrite the item registry fragment and module instructions to require the
|
||||
supplied canonical name and current-chunk evidence without mentioning an
|
||||
ID. Preserve prompt order and cache controls.
|
||||
4. Change the item registry's model projection from ordered `{id,name}` pairs
|
||||
to ordered names-only objects, add a comparison-key index, and expose a
|
||||
defensive `Lookup(name) (dnd.Item, bool)` analogous to the NPC registry.
|
||||
Retain exact `LookupID` for durable normalizers and validators. Because item
|
||||
IDs are derived from the item comparison identity, the names-only
|
||||
`ProjectionDigest` remains sufficient for model input and existing
|
||||
checkpoint consumers.
|
||||
5. Remove `ItemID` from the private response. During response canonicalization,
|
||||
resolve every contextual name, replace it with the registry record's
|
||||
canonical name, and attach its durable ID when constructing the final
|
||||
`dnd.ItemOccurrence`. Unknown selections fail the complete extraction; do
|
||||
not alter evidence or nullable-field validation ownership.
|
||||
6. Change `mappingPolicy` to
|
||||
`dnd.item_occurrences.extract_mapping.v2`.
|
||||
7. Update focused tests to cover names-only projection, defensive comparison
|
||||
lookup, schema rejection of `item_id`, deterministic durable mapping,
|
||||
unknown-name failure after an otherwise valid record, empty registry/result
|
||||
behavior, and preservation of nullable/kind-specific fields.
|
||||
|
||||
### Acceptance Criteria
|
||||
|
||||
- Model-visible item registry and response content contain no item hash.
|
||||
- Every accepted durable occurrence has the matched registry ID and canonical
|
||||
name.
|
||||
- Invalid selection remains all-or-nothing, and existing normalizer/validator
|
||||
defense in depth remains unchanged.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go fmt ./internal/modules/dnd/items/registry ./internal/modules/dnd/extract/itemoccurrences
|
||||
go test ./internal/modules/dnd/items/registry ./internal/modules/dnd/extract/itemoccurrences ./internal/modules/dnd/normalize/itemoccurrences ./internal/modules/dnd/validate/itemoccurrences/...
|
||||
```
|
||||
|
||||
This stage is suitable for one gpt-5.6-terra prompt.
|
||||
|
||||
## Stage 3: Add Contextual Location Grounding
|
||||
|
||||
### Goal
|
||||
|
||||
Replace the location registry's ID/name prompt projection with an immutable,
|
||||
source-aware grounding object that can represent same-name locations safely.
|
||||
Introduce the new path alongside the old occurrence input so this stage remains
|
||||
buildable; Stage 4 performs the atomic extractor cutover and removes the old
|
||||
path.
|
||||
|
||||
### Work
|
||||
|
||||
1. Inspect the location registry, location identity and source-reference
|
||||
helpers, the generic source document index, occurrence checkpoint consumers,
|
||||
and their focused tests.
|
||||
2. In `internal/modules/dnd/locations/registry`, define the private-model
|
||||
types needed by both grounding and the location occurrence extractor:
|
||||
- a returned selector with exactly `name` and `registry_refs`;
|
||||
- a registry projection entry with exactly `name`, `registry_refs`, and
|
||||
`context`;
|
||||
- a source-free range with exactly `start_unit_id` and `end_unit_id`; and
|
||||
- a context unit with exactly `unit_id` and `text`.
|
||||
All fields are required in their private JSON shapes, and constructors and
|
||||
accessors must make defensive copies.
|
||||
3. Add an operation-scoped immutable grounding type constructed from a resolved
|
||||
registry and the current `*source.SourceDocument`. Its API must provide:
|
||||
- a cloned `contracts.LLMInputMaterial` for the `location_registry` slot;
|
||||
- deterministic resolution of a returned selector to one cloned
|
||||
`dnd.Location`.
|
||||
The prompt material's existing `Digest` field owns the digest of the exact
|
||||
model projection; do not expose a second grounding-specific digest API.
|
||||
4. Construct the projection in registry order. Group entries by the existing
|
||||
location comparison key:
|
||||
- every projection entry has exactly the required fields `name`,
|
||||
`registry_refs`, and `context`;
|
||||
- comparison-unique entries use empty `registry_refs` and `context` arrays;
|
||||
- every same-name entry uses its complete canonical source ranges stripped
|
||||
of `source_id` and includes ordered context units covered by those ranges;
|
||||
- each context unit contains exactly required integer `unit_id` and string
|
||||
`text` fields, and units are deduplicated in source order; and
|
||||
- same-name ranges must have `SourceID == doc.ID` and pass
|
||||
`source.DocumentIndex.ValidateRef`.
|
||||
5. Fail grounding construction with a bounded, content-safe error if the
|
||||
source is nil, a same-name range is invalid or belongs to another source,
|
||||
a comparison key is empty, or two records produce the same selector. Do not
|
||||
expose transcript text in the error.
|
||||
6. Resolution uses the existing comparison key. A unique-name selector is
|
||||
accepted only with empty `registry_refs`; a same-name selector is accepted
|
||||
only on an exact canonical range match. Reject unknown names, a non-empty
|
||||
range list for a unique name, an empty/partial/reordered range list for an
|
||||
ambiguous name, or any selector not present in the grounding.
|
||||
7. Separate deterministic identity fingerprinting from LLM material. Add
|
||||
`IdentityDigest()` over the registry's ordered `{id,name}` identity
|
||||
projection, and update the location normalizer and registry-validator
|
||||
checkpoint consumers to use it. The new operation grounding carries the
|
||||
model projection digest in its `LLMInputMaterial`. Retain the old ID-bearing
|
||||
prompt accessor only as a documented transitional dependency of the
|
||||
still-unchanged location occurrence extractor; do not add new callers.
|
||||
8. Add focused tests for unique names, same-name context and selectors,
|
||||
canonical range order, deterministic projection/digest, defensive copies,
|
||||
exact selector resolution, nil/foreign/invalid references, selector
|
||||
collisions, empty registries, and identity fingerprint stability. Do not
|
||||
assert large rendered prompt strings; decode the JSON projection and assert
|
||||
its semantic shape.
|
||||
|
||||
### Acceptance Criteria
|
||||
|
||||
- The location package can build and resolve contextual selectors without
|
||||
exposing `location_id`, `source_id`, digests, or replacement labels.
|
||||
- Same-name locations remain distinct and receive meaningful bounded context.
|
||||
- Deterministic checkpoint consumers retain an ID-sensitive fingerprint.
|
||||
- Only the existing occurrence extractor remains wired to the legacy
|
||||
ID-bearing prompt path until Stage 4; the repository compiles and tests pass.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go fmt ./internal/modules/dnd/locations/registry ./internal/modules/dnd/normalize/locationoccurrences ./internal/modules/dnd/validate/locationoccurrences/...
|
||||
go test ./internal/modules/dnd/locations/registry ./internal/modules/dnd/normalize/locationoccurrences ./internal/modules/dnd/validate/locationoccurrences/...
|
||||
```
|
||||
|
||||
This stage is suitable for one gpt-5.6-terra prompt.
|
||||
|
||||
## Stage 4: Convert Location Occurrences To Contextual Resolution
|
||||
|
||||
### Goal
|
||||
|
||||
Wire the Stage 3 grounding object into location occurrence extraction and
|
||||
remove durable location IDs from the prompt and private response.
|
||||
|
||||
### Work
|
||||
|
||||
1. Inspect `assets/dnd/location-occurrences/`, the location occurrence model,
|
||||
schema loader, extractor, canonicalization, prompt tests, and Stage 3
|
||||
grounding tests.
|
||||
2. Change `dnd_location_occurrences_llm.v1.json` so each occurrence requires
|
||||
exactly `name`, `registry_refs`, `kind`, and `source_refs`; remove
|
||||
`location_id`. Keep all four occurrence kinds. Make `registry_refs` a
|
||||
required array, including an empty array, of strict required positive
|
||||
integer ranges. Keep occurrence `source_refs` separate and unchanged.
|
||||
3. Rewrite `location-registry.md` and module instructions to explain the two
|
||||
selector cases, require exact supplied contextual selectors, prohibit
|
||||
invented locations, and state that registry ranges/context are identity
|
||||
grounding rather than occurrence evidence. Preserve manifest order and
|
||||
cache controls.
|
||||
4. Change the private response type to `Name`, `RegistryRefs`, `Kind`, and
|
||||
`SourceRefs`. Do not reuse `source.SourceRef` for the source-free registry
|
||||
range type.
|
||||
5. In `Extract`, construct operation grounding from the resolved registry and
|
||||
`req.Source` before calling the LLM, put its projection in the
|
||||
`location_registry` input, and resolve every returned selector after
|
||||
completion. Attach the selected registry record's exact ID and canonical
|
||||
name to the durable occurrence while retaining only the response's
|
||||
current-source `source_refs` as evidence.
|
||||
6. Fail the whole extraction on grounding-construction failure or the first
|
||||
unknown, malformed, mismatched, or ambiguous selector. This replaces the
|
||||
current behavior that can preserve unknown ID/name pairs for later
|
||||
validators. Keep later normalizer and validator checks as defense in depth
|
||||
for artifacts entering other boundaries.
|
||||
7. Change `mappingPolicy` to
|
||||
`dnd.location_occurrences.extract_mapping.v2`.
|
||||
8. Remove the legacy ID-bearing registry `PromptInput` and its model-projection
|
||||
digest once the extractor uses operation grounding. Keep the occurrence's
|
||||
static module fingerprint based on `IdentityDigest()`, prompt/schema
|
||||
fingerprints, and mapping policy. The operation-scoped projection is already
|
||||
covered by source/chunk identity and configured or generated reference
|
||||
dependencies, while its `LLMInputMaterial.Digest` identifies the exact model
|
||||
input; do not add a second digest API or operation-aware static fingerprint.
|
||||
9. Update focused schema, prompt, extractor, canonicalization, checkpoint, and
|
||||
generated-reference tests. Cover unique-name empty selectors, successful
|
||||
same-name selection, failure for an unsupported ambiguous mention,
|
||||
partial/reordered ranges, no registry-to-occurrence evidence leakage,
|
||||
all-or-nothing failure, empty registry/result behavior, and unchanged
|
||||
durable ordering/deduplication.
|
||||
|
||||
### Acceptance Criteria
|
||||
|
||||
- The location prompt and private response contain no durable location ID.
|
||||
- Unique and same-name records resolve according to the final selector
|
||||
contract.
|
||||
- Accepted durable output is unchanged in shape and still contains an exact
|
||||
location ID/name pair.
|
||||
- Registry context cannot become durable occurrence evidence.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go fmt ./internal/modules/dnd/extract/locationoccurrences ./internal/modules/dnd/locations/registry
|
||||
go test ./internal/modules/dnd/locations/registry ./internal/modules/dnd/extract/locationoccurrences ./internal/modules/dnd/normalize/locationoccurrences ./internal/modules/dnd/validate/locationoccurrences/...
|
||||
```
|
||||
|
||||
This stage is suitable for one gpt-5.6-terra prompt.
|
||||
|
||||
## Stage 5: Replace Reconciliation Keys With Contextual Descriptors
|
||||
|
||||
### Goal
|
||||
|
||||
Change the shared NPC/item/location registry-normalization proposal contract so
|
||||
opaque candidate keys remain internal and the model sees and returns only
|
||||
names plus evidence coordinates.
|
||||
|
||||
### Work
|
||||
|
||||
1. Inspect:
|
||||
- `internal/modules/dnd/shared/entityreconcile/`;
|
||||
- `assets/dnd/shared/prompts/common-dnd-entity-reconciliation.md`;
|
||||
- `assets/dnd/entity-reconciliation/schemas/`;
|
||||
- all three registry normalization manifests and prompt tests; and
|
||||
- the NPC, item, and location registry normalizers and reconciliation tests.
|
||||
2. Introduce one exported, defensively copied contextual selector type in
|
||||
`entityreconcile` with JSON `name` and `source_refs`, plus a strict
|
||||
source-free range type. Use it for candidate input views and for
|
||||
`DuplicateGroup.Members` and `.Canonical`.
|
||||
3. Keep deterministic candidate keys only inside `Materials`. During
|
||||
`BuildContext`, validate and canonicalize candidate references as today,
|
||||
serialize candidate views without `key`, derive a stable internal lookup
|
||||
from canonical selector JSON to the corresponding internal candidate key,
|
||||
and detect descriptor collisions before eligibility is established.
|
||||
Colliding candidates must not appear in the prompt input or become
|
||||
eligible; their records remain in deterministic normalization output.
|
||||
4. Update `Materials.Assess` to resolve every returned selector through that
|
||||
internal lookup before running the existing group assessment. Preserve
|
||||
existing issue categories where their meaning still applies. Treat an
|
||||
unknown or collided descriptor as an unknown/ineligible selection, discard
|
||||
only the affected group, and retain existing overlap handling. `SafeGroup`
|
||||
may continue returning internal candidate keys so the three domain
|
||||
normalizers retain their position mapping; those keys are not model-facing.
|
||||
5. Rewrite `dnd_entity_reconcile_llm.v1.json` so members and canonical are
|
||||
strict selector objects. Require non-empty `name` structurally where the
|
||||
current schemas do so, require `source_refs`, and make each range strict
|
||||
with required positive integer endpoints. Preserve `duplicate_groups` and
|
||||
the existing semantic assessment of minimum group size, membership,
|
||||
duplicates, eligibility, and overlap rather than moving every semantic
|
||||
failure into JSON Schema.
|
||||
6. Rewrite the shared reconciliation fragment to tell the model to return
|
||||
supplied contextual descriptors and never invent names or ranges. Remove
|
||||
every instruction about opaque keys. Preserve all three manifests' message
|
||||
ordering and cache controls.
|
||||
7. Change registry normalization policy identifiers to:
|
||||
- `dnd.npc_registry.normalize.v4`;
|
||||
- `dnd.item_registry.normalize.v2`; and
|
||||
- `dnd.location_registry.normalize.v2`.
|
||||
8. Update shared and domain tests to cover candidate JSON without keys,
|
||||
contextual proposal decoding, valid selector-to-internal-key mapping,
|
||||
equal names with different evidence, descriptor collision exclusion,
|
||||
unknown/partial/reordered descriptors, overlapping groups, canonical
|
||||
membership, invalid structured-output fallback, currency safety, and
|
||||
preservation of every non-applied deterministic candidate. Update prompt
|
||||
asset fixtures to the new selector schema; do not snapshot prompt prose.
|
||||
|
||||
### Acceptance Criteria
|
||||
|
||||
- No registry normalization prompt input or private response contains a
|
||||
`candidate-*` key.
|
||||
- Internal keys remain inaccessible to the model but may still support safe
|
||||
deterministic position mapping.
|
||||
- All existing normalizer safety, retry, fallback, warning, currency, and
|
||||
same-name-location policies remain intact.
|
||||
- Identical contextual descriptors cannot be arbitrarily reconciled.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go fmt ./internal/modules/dnd/shared/entityreconcile ./internal/modules/dnd/normalize/npcregistry ./internal/modules/dnd/normalize/itemregistry ./internal/modules/dnd/normalize/locationregistry
|
||||
go test ./internal/modules/dnd/shared/entityreconcile ./internal/modules/dnd/normalize/npcregistry ./internal/modules/dnd/normalize/itemregistry ./internal/modules/dnd/normalize/locationregistry
|
||||
```
|
||||
|
||||
This is the largest stage, but it is one cohesive shared-contract migration
|
||||
and is suitable for one gpt-5.6-terra prompt when implemented exactly within
|
||||
the listed packages. Do not combine it with occurrence or documentation work.
|
||||
|
||||
## Stage 6: Record The Decision And Update Canonical Documentation
|
||||
|
||||
### Goal
|
||||
|
||||
Document the implemented policy in its durable architectural, internal, and
|
||||
integration homes without duplicating volatile details or presenting roadmap
|
||||
work as current behavior prematurely.
|
||||
|
||||
### Work
|
||||
|
||||
1. Re-read `docs/policy/documentation.md`, ADR-0003, ADR-0009, ADR-0011,
|
||||
`docs/internal/dnd.md`, `docs/internal/llm.md`, and the six affected registry
|
||||
and occurrence integration documents. Verify the code before describing it.
|
||||
2. Add
|
||||
`docs/adr/0012-resolve-opaque-entity-identifiers-deterministically.md` in
|
||||
the repository's Nygard ADR format with status `Accepted` and the actual
|
||||
implementation date. Record the model-semantic/deterministic-identity
|
||||
boundary, source-coordinate allowance, request-local-label exception,
|
||||
alternatives, ambiguity behavior, and consequences. Link ADR-0003 and
|
||||
ADR-0009 rather than repeating their complete decisions.
|
||||
3. Add a concise normative invariant under the LLM boundary in
|
||||
`docs/policy/architecture.md`: callers use contextual model selections and
|
||||
attach opaque application identities deterministically when possible. Link
|
||||
ADR-0012 for rationale.
|
||||
4. Update `docs/internal/dnd.md` to replace exact model-facing `{id,name}`
|
||||
claims with the implemented NPC/item names-only and location contextual
|
||||
selector behavior. Document reconciliation descriptors, internal-only keys,
|
||||
all-or-nothing occurrence mapping failures, normalization fallback, and the
|
||||
separation between registry and occurrence evidence. Do not duplicate the
|
||||
private JSON schemas.
|
||||
5. Add only a short ownership clarification to `docs/internal/llm.md`: the
|
||||
calling module resolves contextual selections; PromptKit and its adapter do
|
||||
not own entity identity.
|
||||
6. Update these durable integration contracts while preserving their public
|
||||
ID-bearing wire examples and schema statements:
|
||||
- `docs/integrations/dnd-npc-registry-artifacts.md`;
|
||||
- `docs/integrations/dnd-npc-occurrence-artifacts.md`;
|
||||
- `docs/integrations/dnd-item-registry-artifacts.md`;
|
||||
- `docs/integrations/dnd-item-occurrence-artifacts.md`;
|
||||
- `docs/integrations/dnd-location-registry-artifacts.md`; and
|
||||
- `docs/integrations/dnd-location-occurrence-artifacts.md`.
|
||||
Remove claims that LLM consumers receive `{id,name}` or that raw model
|
||||
output supplies an ID. State that Notarius maps contextual output into the
|
||||
unchanged exact durable pair.
|
||||
7. Revise the generic LLM-assisted deduplication entry in
|
||||
`docs/roadmap/future.md`: stable unique IDs remain internal deterministic
|
||||
state, while a future model proposal uses contextual descriptors or a
|
||||
specifically justified request-local short label.
|
||||
8. Do not change README, CLI, configuration, operations, examples, or public
|
||||
schema files; this feature has no user-selectable surface or public wire
|
||||
change.
|
||||
|
||||
### Acceptance Criteria
|
||||
|
||||
- ADR-0012 owns rationale; architecture owns the normative boundary; internal
|
||||
docs own mechanics; integration docs own unchanged durable contracts; and
|
||||
the future roadmap no longer proposes durable IDs as the default model
|
||||
selector.
|
||||
- No current-behavior document claims that a model copies hash-based entity
|
||||
IDs or opaque reconciliation keys.
|
||||
- Documentation does not duplicate private schemas or implementation history.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
git diff --check
|
||||
rg -n '\{id,name\}|ID/name grounding|Candidate keys are opaque|candidate-[0-9]' docs assets/dnd
|
||||
```
|
||||
|
||||
Review every search result semantically; durable wire-contract ID/name
|
||||
requirements and internal test fixtures are not automatically errors.
|
||||
|
||||
This stage is suitable for one gpt-5.6-terra prompt.
|
||||
|
||||
## Stage 7: Integration Audit And Final Verification
|
||||
|
||||
### Goal
|
||||
|
||||
Verify the assembled D&D family, remove obsolete identity-copy paths, and
|
||||
finish with a clean, policy-compliant implementation.
|
||||
|
||||
### Work
|
||||
|
||||
1. Audit every maintained D&D prompt manifest, selected fragment, private
|
||||
schema, and constructed prompt projection. Confirm that no model is asked to
|
||||
reproduce `npc:sha256:...`, `item:sha256:...`,
|
||||
`location:sha256:...`, `candidate-*`, a UUID, a digest, or another opaque
|
||||
entity handle. Do not confuse runtime metadata or durable output contracts
|
||||
with model-visible material.
|
||||
2. Trace all former APIs and fields, including `IdentityPromptInput`,
|
||||
ID-bearing item/location prompt projections, private `NPCID`/`ItemID`/
|
||||
`LocationID` response fields, and model-visible candidate keys. Remove dead
|
||||
code, obsolete comments, stale test names, and unused assets. Retain
|
||||
identity-only digests and exact durable lookup APIs used by deterministic
|
||||
consumers.
|
||||
3. Review prompt fingerprint registration and checkpoint fingerprints. Confirm
|
||||
that each affected prompt/schema/policy/projection change invalidates the
|
||||
relevant operation and that unrelated D&D lanes retain their existing
|
||||
fingerprints.
|
||||
4. Run representative production registration and multi-step pipeline tests
|
||||
using existing fakes. Update only tests whose stable behavior changed.
|
||||
Confirm generated NPC/item/location registry handoffs still prepare and
|
||||
that final durable occurrences encode and validate under their existing
|
||||
`v1` codecs.
|
||||
5. Run formatting, focused suites, full tests, vet, build, and documentation
|
||||
whitespace checks. Fix only failures caused by this feature. Report any
|
||||
unrelated pre-existing failure without broadening scope.
|
||||
6. Review the feature roadmap acceptance criteria one by one. Do not delete
|
||||
`contextual-entity-grounding.md` or this implementation plan in this stage;
|
||||
roadmap retirement is a separate maintainer action after review.
|
||||
|
||||
### Acceptance Criteria
|
||||
|
||||
- All feature-roadmap acceptance criteria are met.
|
||||
- The repository contains no obsolete model-facing opaque-identity path.
|
||||
- Public artifacts and generated handoffs remain compatible.
|
||||
- Tests are focused on behavior rather than prose or implementation shape.
|
||||
- The worktree contains only intentional feature and documentation changes.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go fmt ./internal/modules/dnd/...
|
||||
go test ./internal/modules/dnd/...
|
||||
go test ./internal/modules/integration/...
|
||||
go test ./...
|
||||
go vet ./...
|
||||
go build ./cmd/notarius
|
||||
git diff --check
|
||||
git status --short
|
||||
```
|
||||
|
||||
This stage is suitable for one gpt-5.6-terra prompt.
|
||||
@@ -189,10 +189,18 @@ func TestMaintainedCompleteExamplePublishesRegistryBackedEntityOccurrences(t *te
|
||||
}
|
||||
|
||||
evidence := readProductionJSON[evidencecontext.Document](t, filepath.Join(runRoot, "evidence-context.json"))
|
||||
for _, laneID := range []string{"enemy-events", "npc-registry", "npc-occurrences", "item-registry", "item-occurrences", "location-registry", "location-occurrences"} {
|
||||
if !containsString(evidence.SelectedLanes, laneID) || !evidenceHasLane(evidence, laneID) {
|
||||
t.Fatalf("evidence context = %#v, want direct %s evidence", evidence, laneID)
|
||||
if len(evidence) == 0 {
|
||||
t.Fatalf("evidence context = %#v, want selected source-unit evidence", evidence)
|
||||
}
|
||||
seenEvidenceUnits := make(map[int]struct{}, len(evidence))
|
||||
for _, unit := range evidence {
|
||||
if unit.Ref.SourceID != "session-ravenfall" || unit.Ref.StartUnitID != unit.ID || unit.Ref.EndUnitID != unit.ID {
|
||||
t.Fatalf("evidence unit = %#v, want unchanged source-unit self-reference", unit)
|
||||
}
|
||||
if _, exists := seenEvidenceUnits[unit.ID]; exists {
|
||||
t.Fatalf("evidence context = %#v, want each source unit once", evidence)
|
||||
}
|
||||
seenEvidenceUnits[unit.ID] = struct{}{}
|
||||
}
|
||||
|
||||
requests := client.requestsFor(enemyevents.PromptID)
|
||||
@@ -415,17 +423,6 @@ func containsString(values []string, want string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
func evidenceHasLane(value evidencecontext.Document, laneID string) bool {
|
||||
for _, context := range value.Contexts {
|
||||
for _, reference := range context.EvidenceRefs {
|
||||
if reference.LaneID == laneID {
|
||||
return true
|
||||
}
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func generatedReferenceBinding(bindings []pipeline.ReferenceBinding, slotName string) (pipeline.ReferenceBinding, bool) {
|
||||
for _, binding := range bindings {
|
||||
if binding.SlotName == slotName && binding.Artifact != nil {
|
||||
|
||||
@@ -25,6 +25,7 @@ import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/chunk/scenes"
|
||||
combatcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/combatturns"
|
||||
@@ -311,6 +312,47 @@ func TestProductionPromptAssetsPrepareWithoutProviderCredentials(t *testing.T) {
|
||||
if _, err := pipeline.Prepare(effective.ResolvedPipeline, components.registries, pipeline.ModuleDependencies{LLM: &productionFakeLLMClient{}}); err != nil {
|
||||
t.Fatalf("prepare production scene and spell modules: %v", err)
|
||||
}
|
||||
|
||||
schemaFS, err := components.assets.SchemaFS()
|
||||
if err != nil {
|
||||
t.Fatalf("production schema assets: %v", err)
|
||||
}
|
||||
if _, err := fs.ReadFile(schemaFS, filepath.Base(semanticreconcile.SchemaAssetPath)); err != nil {
|
||||
t.Fatalf("generic reconciliation schema asset: %v", err)
|
||||
}
|
||||
options, err := components.assets.PromptKitOptions()
|
||||
if err != nil {
|
||||
t.Fatalf("production PromptKit options: %v", err)
|
||||
}
|
||||
options = append(options, promptkit.WithProfiles(promptkit.OpenAICompatibleProfile(promptkit.OpenAICompatibleProfileConfig{
|
||||
ID: "assembled-prompt-test", Endpoint: "http://127.0.0.1:1/v1", Model: "test",
|
||||
})))
|
||||
engine, err := promptkit.NewEngine(promptkit.Config{}, options...)
|
||||
if err != nil {
|
||||
t.Fatalf("production prompt engine: %v", err)
|
||||
}
|
||||
inputs := map[string]promptkit.ArtifactRef{
|
||||
"candidates": promptkit.Inline(`{"candidates":[{"candidate_id":1,"label":"Alias","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`),
|
||||
"transcript": promptkit.Inline(`{"windows":[{"units":[]}]}`),
|
||||
}
|
||||
for _, prompt := range []struct {
|
||||
id string
|
||||
version string
|
||||
}{
|
||||
{id: npcnormalize.PromptID, version: npcnormalize.PromptVersion},
|
||||
{id: itemregistrynormalize.PromptID, version: itemregistrynormalize.PromptVersion},
|
||||
{id: locationnormalize.PromptID, version: locationnormalize.PromptVersion},
|
||||
} {
|
||||
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{
|
||||
PromptID: prompt.id, PromptVersion: prompt.version, ProfileID: "assembled-prompt-test", Inputs: inputs,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("prepare production prompt %q: %v", prompt.id, err)
|
||||
}
|
||||
if prepared.OutputContract.SchemaPath != filepath.Base(semanticreconcile.SchemaAssetPath) {
|
||||
t.Fatalf("prompt %q schema = %q, want generic reconciliation schema", prompt.id, prepared.OutputContract.SchemaPath)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestProductionSpellValidatorsPrepareFromMaterializedCatalog(t *testing.T) {
|
||||
|
||||
@@ -2,24 +2,8 @@
|
||||
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
||||
"$id": "notarius.source.evidence_context",
|
||||
"title": "notarius_source_evidence_context_v1",
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["source_id", "source_digest", "window_units", "selected_lanes", "contexts"],
|
||||
"properties": {
|
||||
"source_id": {"type": "string", "minLength": 1},
|
||||
"source_digest": {"type": "string", "pattern": "^sha256:[0-9a-f]{64}$"},
|
||||
"window_units": {"type": "integer", "minimum": 0},
|
||||
"selected_lanes": {
|
||||
"type": "array",
|
||||
"minItems": 1,
|
||||
"uniqueItems": true,
|
||||
"items": {"type": "string", "minLength": 1}
|
||||
},
|
||||
"contexts": {
|
||||
"type": "array",
|
||||
"items": {"$ref": "#/$defs/context"}
|
||||
}
|
||||
},
|
||||
"type": "array",
|
||||
"items": {"$ref": "#/$defs/unit"},
|
||||
"$defs": {
|
||||
"source_ref": {
|
||||
"type": "object",
|
||||
@@ -42,25 +26,6 @@
|
||||
"ref": {"$ref": "#/$defs/source_ref"},
|
||||
"metadata": {"type": "object", "additionalProperties": true}
|
||||
}
|
||||
},
|
||||
"evidence_ref": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["lane_id", "source_ref"],
|
||||
"properties": {
|
||||
"lane_id": {"type": "string", "minLength": 1},
|
||||
"source_ref": {"$ref": "#/$defs/source_ref"}
|
||||
}
|
||||
},
|
||||
"context": {
|
||||
"type": "object",
|
||||
"additionalProperties": false,
|
||||
"required": ["context_ref", "evidence_refs", "units"],
|
||||
"properties": {
|
||||
"context_ref": {"$ref": "#/$defs/source_ref"},
|
||||
"evidence_refs": {"type": "array", "minItems": 1, "items": {"$ref": "#/$defs/evidence_ref"}},
|
||||
"units": {"type": "array", "minItems": 1, "items": {"$ref": "#/$defs/unit"}}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,118 +3,62 @@ package evidencecontext
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
)
|
||||
|
||||
type contribution struct {
|
||||
laneID string
|
||||
ref source.SourceRef
|
||||
type expandedRange struct {
|
||||
startPos int
|
||||
endPos int
|
||||
}
|
||||
|
||||
type expandedRange struct {
|
||||
startPos int
|
||||
endPos int
|
||||
contributions []contribution
|
||||
}
|
||||
|
||||
// Build validates accepted direct references, expands them by source-document
|
||||
// position, and returns their deterministic context union.
|
||||
// Build validates projected source references, expands them by source-document
|
||||
// position, and returns their ordered union as an owned source-unit excerpt.
|
||||
func Build(request BuildRequest) (Document, error) {
|
||||
if request.WindowUnits < 0 {
|
||||
return Document{}, fmt.Errorf("window_units must not be negative")
|
||||
}
|
||||
lanes, err := normalizeSelectedLanes(request.SelectedLanes)
|
||||
if err != nil {
|
||||
return Document{}, err
|
||||
return nil, fmt.Errorf("window_units must not be negative")
|
||||
}
|
||||
if err := source.ValidateDocument(request.Source); err != nil {
|
||||
return Document{}, fmt.Errorf("validate source document: %w", err)
|
||||
return nil, fmt.Errorf("validate source document: %w", err)
|
||||
}
|
||||
digest, err := source.DigestDocument(request.Source)
|
||||
if err != nil {
|
||||
return Document{}, fmt.Errorf("digest source document: %w", err)
|
||||
return nil, fmt.Errorf("digest source document: %w", err)
|
||||
}
|
||||
if digest != request.Source.Digest {
|
||||
return Document{}, fmt.Errorf("source digest does not match source document digest")
|
||||
return nil, fmt.Errorf("source digest does not match source document digest")
|
||||
}
|
||||
|
||||
selected := make(map[string]struct{}, len(lanes))
|
||||
for _, laneID := range lanes {
|
||||
selected[laneID] = struct{}{}
|
||||
}
|
||||
index := source.NewDocumentIndex(request.Source)
|
||||
seen := make(map[evidenceKey]struct{})
|
||||
contributions := make([]contribution, 0)
|
||||
for laneIndex, laneEvidence := range request.LaneEvidence {
|
||||
laneID := strings.TrimSpace(laneEvidence.LaneID)
|
||||
if _, ok := selected[laneID]; !ok {
|
||||
return Document{}, fmt.Errorf("lane evidence[%d] lane %q is not selected", laneIndex, laneID)
|
||||
ranges := make([]expandedRange, 0, len(request.SourceRefs))
|
||||
for refIndex, ref := range request.SourceRefs {
|
||||
if err := index.ValidateRef(ref); err != nil {
|
||||
return nil, fmt.Errorf("source reference[%d]: %w", refIndex, err)
|
||||
}
|
||||
for refIndex, ref := range laneEvidence.SourceRefs {
|
||||
if err := index.ValidateRef(ref); err != nil {
|
||||
return Document{}, fmt.Errorf("lane %q source reference[%d]: %w", laneID, refIndex, err)
|
||||
startPos, _ := index.Position(ref.StartUnitID)
|
||||
endPos, _ := index.Position(ref.EndUnitID)
|
||||
ranges = append(ranges, expandedRange{
|
||||
startPos: expandStart(startPos, request.WindowUnits),
|
||||
endPos: expandEnd(endPos, len(request.Source.Units), request.WindowUnits),
|
||||
})
|
||||
}
|
||||
|
||||
merged := mergeRanges(ranges)
|
||||
unitCount := 0
|
||||
for _, value := range merged {
|
||||
unitCount += value.endPos - value.startPos + 1
|
||||
}
|
||||
document := make(Document, 0, unitCount)
|
||||
for _, value := range merged {
|
||||
for position := value.startPos; position <= value.endPos; position++ {
|
||||
unit, err := cloneSourceUnit(request.Source.Units[position])
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("clone source unit at position %d: %w", position, err)
|
||||
}
|
||||
key := evidenceKey{laneID: laneID, ref: ref}
|
||||
if _, exists := seen[key]; exists {
|
||||
continue
|
||||
}
|
||||
seen[key] = struct{}{}
|
||||
startPos, _ := index.Position(ref.StartUnitID)
|
||||
endPos, _ := index.Position(ref.EndUnitID)
|
||||
contributions = append(contributions, contribution{laneID: laneID, ref: ref, startPos: expandStart(startPos, request.WindowUnits), endPos: expandEnd(endPos, len(request.Source.Units), request.WindowUnits)})
|
||||
document = append(document, unit)
|
||||
}
|
||||
}
|
||||
|
||||
sort.Slice(contributions, func(i, j int) bool { return lessContribution(contributions[i], contributions[j]) })
|
||||
document := Document{
|
||||
SourceID: request.Source.ID,
|
||||
SourceDigest: digest,
|
||||
WindowUnits: request.WindowUnits,
|
||||
SelectedLanes: lanes,
|
||||
Contexts: make([]Context, 0),
|
||||
}
|
||||
for _, rangeValue := range mergeRanges(contributions) {
|
||||
context, err := buildContext(request.Source, rangeValue)
|
||||
if err != nil {
|
||||
return Document{}, err
|
||||
}
|
||||
document.Contexts = append(document.Contexts, context)
|
||||
}
|
||||
canonical, err := canonicalizeOwned(document)
|
||||
if err != nil {
|
||||
return Document{}, fmt.Errorf("validate evidence context: %w", err)
|
||||
}
|
||||
return canonical, nil
|
||||
}
|
||||
|
||||
type evidenceKey struct {
|
||||
laneID string
|
||||
ref source.SourceRef
|
||||
}
|
||||
|
||||
func normalizeSelectedLanes(values []string) ([]string, error) {
|
||||
if len(values) == 0 {
|
||||
return nil, fmt.Errorf("selected_lanes must not be empty")
|
||||
}
|
||||
seen := make(map[string]struct{}, len(values))
|
||||
lanes := make([]string, 0, len(values))
|
||||
for index, raw := range values {
|
||||
laneID := strings.TrimSpace(raw)
|
||||
if laneID == "" {
|
||||
return nil, fmt.Errorf("selected_lanes[%d] must not be empty", index)
|
||||
}
|
||||
if _, exists := seen[laneID]; exists {
|
||||
return nil, fmt.Errorf("selected_lanes lane %q is duplicated", laneID)
|
||||
}
|
||||
seen[laneID] = struct{}{}
|
||||
lanes = append(lanes, laneID)
|
||||
}
|
||||
sort.Strings(lanes)
|
||||
return lanes, nil
|
||||
return document, nil
|
||||
}
|
||||
|
||||
func expandStart(position, window int) int {
|
||||
@@ -132,65 +76,25 @@ func expandEnd(position, length, window int) int {
|
||||
return position + window
|
||||
}
|
||||
|
||||
func lessContribution(left, right contribution) bool {
|
||||
if left.startPos != right.startPos {
|
||||
return left.startPos < right.startPos
|
||||
}
|
||||
if left.endPos != right.endPos {
|
||||
return left.endPos < right.endPos
|
||||
}
|
||||
return lessEvidenceRef(EvidenceRef{LaneID: left.laneID, SourceRef: left.ref}, EvidenceRef{LaneID: right.laneID, SourceRef: right.ref})
|
||||
}
|
||||
|
||||
func mergeRanges(values []contribution) []expandedRange {
|
||||
func mergeRanges(values []expandedRange) []expandedRange {
|
||||
if len(values) == 0 {
|
||||
return nil
|
||||
}
|
||||
ranges := make([]expandedRange, 0, len(values))
|
||||
sort.Slice(values, func(i, j int) bool {
|
||||
if values[i].startPos != values[j].startPos {
|
||||
return values[i].startPos < values[j].startPos
|
||||
}
|
||||
return values[i].endPos < values[j].endPos
|
||||
})
|
||||
merged := make([]expandedRange, 0, len(values))
|
||||
for _, value := range values {
|
||||
if len(ranges) == 0 || value.startPos > ranges[len(ranges)-1].endPos+1 {
|
||||
ranges = append(ranges, expandedRange{startPos: value.startPos, endPos: value.endPos, contributions: []contribution{value}})
|
||||
if len(merged) == 0 || value.startPos > merged[len(merged)-1].endPos+1 {
|
||||
merged = append(merged, value)
|
||||
continue
|
||||
}
|
||||
current := &ranges[len(ranges)-1]
|
||||
if value.endPos > current.endPos {
|
||||
current.endPos = value.endPos
|
||||
if value.endPos > merged[len(merged)-1].endPos {
|
||||
merged[len(merged)-1].endPos = value.endPos
|
||||
}
|
||||
current.contributions = append(current.contributions, value)
|
||||
}
|
||||
return ranges
|
||||
}
|
||||
|
||||
func buildContext(document *source.SourceDocument, value expandedRange) (Context, error) {
|
||||
evidenceRefs := make([]EvidenceRef, 0, len(value.contributions))
|
||||
for _, contribution := range value.contributions {
|
||||
evidenceRefs = append(evidenceRefs, EvidenceRef{LaneID: contribution.laneID, SourceRef: contribution.ref})
|
||||
}
|
||||
sort.Slice(evidenceRefs, func(i, j int) bool { return lessEvidenceRef(evidenceRefs[i], evidenceRefs[j]) })
|
||||
units := make([]source.SourceUnit, 0, value.endPos-value.startPos+1)
|
||||
for position := value.startPos; position <= value.endPos; position++ {
|
||||
unit, err := cloneSourceUnit(document.Units[position])
|
||||
if err != nil {
|
||||
return Context{}, fmt.Errorf("clone source unit at position %d: %w", position, err)
|
||||
}
|
||||
units = append(units, unit)
|
||||
}
|
||||
return Context{
|
||||
ContextRef: source.SourceRef{SourceID: document.ID, StartUnitID: units[0].ID, EndUnitID: units[len(units)-1].ID},
|
||||
EvidenceRefs: evidenceRefs,
|
||||
Units: units,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func lessEvidenceRef(left, right EvidenceRef) bool {
|
||||
if left.LaneID != right.LaneID {
|
||||
return left.LaneID < right.LaneID
|
||||
}
|
||||
if left.SourceRef.SourceID != right.SourceRef.SourceID {
|
||||
return left.SourceRef.SourceID < right.SourceRef.SourceID
|
||||
}
|
||||
if left.SourceRef.StartUnitID != right.SourceRef.StartUnitID {
|
||||
return left.SourceRef.StartUnitID < right.SourceRef.StartUnitID
|
||||
}
|
||||
return left.SourceRef.EndUnitID < right.SourceRef.EndUnitID
|
||||
return merged
|
||||
}
|
||||
|
||||
@@ -6,7 +6,6 @@ import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"regexp"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
@@ -18,8 +17,6 @@ import (
|
||||
//go:embed assets/schemas/source_evidence_context.v1.json
|
||||
var schemaAssets embed.FS
|
||||
|
||||
var digestPattern = regexp.MustCompile(`^sha256:[0-9a-f]{64}$`)
|
||||
|
||||
var (
|
||||
loadSchemaOnce sync.Once
|
||||
loadedSchema []byte
|
||||
@@ -78,24 +75,24 @@ func (c *Codec) Encode(value Document) ([]byte, error) {
|
||||
|
||||
func (c *Codec) Decode(content []byte) (Document, error) {
|
||||
if _, err := c.schemaBytes(); err != nil {
|
||||
return Document{}, err
|
||||
return nil, err
|
||||
}
|
||||
if err := validateSchemaInstance(content); err != nil {
|
||||
return Document{}, fmt.Errorf("decode evidence context: %w", err)
|
||||
return nil, fmt.Errorf("decode evidence context: %w", err)
|
||||
}
|
||||
decoder := json.NewDecoder(bytes.NewReader(content))
|
||||
decoder.DisallowUnknownFields()
|
||||
var value Document
|
||||
if err := decoder.Decode(&value); err != nil {
|
||||
return Document{}, fmt.Errorf("decode evidence context: %w", err)
|
||||
return nil, fmt.Errorf("decode evidence context: %w", err)
|
||||
}
|
||||
var trailing any
|
||||
if err := decoder.Decode(&trailing); err != io.EOF {
|
||||
return Document{}, fmt.Errorf("decode evidence context: multiple JSON values")
|
||||
return nil, fmt.Errorf("decode evidence context: multiple JSON values")
|
||||
}
|
||||
canonical, err := canonicalizeOwned(value)
|
||||
if err != nil {
|
||||
return Document{}, fmt.Errorf("decode evidence context: %w", err)
|
||||
return nil, fmt.Errorf("decode evidence context: %w", err)
|
||||
}
|
||||
return canonical, nil
|
||||
}
|
||||
@@ -115,17 +112,16 @@ func loadAndCompileSchema() {
|
||||
return
|
||||
}
|
||||
var identity struct {
|
||||
ID string `json:"$id"`
|
||||
Title string `json:"title"`
|
||||
Type string `json:"type"`
|
||||
Required []string `json:"required"`
|
||||
ID string `json:"$id"`
|
||||
Title string `json:"title"`
|
||||
Type string `json:"type"`
|
||||
}
|
||||
if err := json.Unmarshal(raw, &identity); err != nil {
|
||||
loadSchemaErr = fmt.Errorf("decode source evidence context schema: %w", err)
|
||||
return
|
||||
}
|
||||
if identity.ID != SchemaID || identity.Title != SchemaName || identity.Type != "object" || !hasRequiredFields(identity.Required) {
|
||||
loadSchemaErr = fmt.Errorf("source evidence context schema identity or required fields are invalid")
|
||||
if identity.ID != SchemaID || identity.Title != SchemaName || identity.Type != "array" {
|
||||
loadSchemaErr = fmt.Errorf("source evidence context schema identity is invalid")
|
||||
return
|
||||
}
|
||||
schemaDocument, err := jsonschema.UnmarshalJSON(bytes.NewReader(raw))
|
||||
@@ -158,164 +154,65 @@ func validateSchemaInstance(content []byte) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
func hasRequiredFields(required []string) bool {
|
||||
want := map[string]bool{"source_id": true, "source_digest": true, "window_units": true, "selected_lanes": true, "contexts": true}
|
||||
for _, field := range required {
|
||||
delete(want, field)
|
||||
}
|
||||
return len(want) == 0
|
||||
}
|
||||
|
||||
func canonicalize(value Document) (Document, error) {
|
||||
owned, err := clone(value)
|
||||
if err != nil {
|
||||
return Document{}, err
|
||||
return nil, err
|
||||
}
|
||||
return canonicalizeOwned(owned)
|
||||
}
|
||||
|
||||
func canonicalizeOwned(value Document) (Document, error) {
|
||||
if err := requireIdentity("source_id", value.SourceID); err != nil {
|
||||
return Document{}, err
|
||||
if value == nil {
|
||||
return nil, fmt.Errorf("document must be a JSON array")
|
||||
}
|
||||
if !digestPattern.MatchString(value.SourceDigest) {
|
||||
return Document{}, fmt.Errorf("source_digest must be a sha256 digest")
|
||||
}
|
||||
if value.WindowUnits < 0 {
|
||||
return Document{}, fmt.Errorf("window_units must not be negative")
|
||||
}
|
||||
if err := validateSelectedLanes(value.SelectedLanes); err != nil {
|
||||
return Document{}, err
|
||||
}
|
||||
if value.Contexts == nil {
|
||||
value.Contexts = make([]Context, 0)
|
||||
}
|
||||
selected := make(map[string]struct{}, len(value.SelectedLanes))
|
||||
for _, laneID := range value.SelectedLanes {
|
||||
selected[laneID] = struct{}{}
|
||||
}
|
||||
seenUnits := make(map[int]struct{})
|
||||
for contextIndex := range value.Contexts {
|
||||
context, err := canonicalizeContext(value.SourceID, selected, seenUnits, value.Contexts[contextIndex], contextIndex)
|
||||
if err != nil {
|
||||
return Document{}, err
|
||||
}
|
||||
value.Contexts[contextIndex] = context
|
||||
}
|
||||
return value, nil
|
||||
}
|
||||
|
||||
func validateSelectedLanes(lanes []string) error {
|
||||
if len(lanes) == 0 {
|
||||
return fmt.Errorf("selected_lanes must not be empty")
|
||||
}
|
||||
for index, laneID := range lanes {
|
||||
if err := requireIdentity(fmt.Sprintf("selected_lanes[%d]", index), laneID); err != nil {
|
||||
return err
|
||||
}
|
||||
if index > 0 && lanes[index-1] >= laneID {
|
||||
return fmt.Errorf("selected_lanes must be unique and in lexical order")
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func canonicalizeContext(sourceID string, selected map[string]struct{}, seenUnits map[int]struct{}, value Context, contextIndex int) (Context, error) {
|
||||
prefix := fmt.Sprintf("contexts[%d]", contextIndex)
|
||||
if len(value.EvidenceRefs) == 0 {
|
||||
return Context{}, fmt.Errorf("%s.evidence_refs must not be empty", prefix)
|
||||
}
|
||||
if len(value.Units) == 0 {
|
||||
return Context{}, fmt.Errorf("%s.units must not be empty", prefix)
|
||||
}
|
||||
if err := validateRefIdentity(sourceID, value.ContextRef, prefix+".context_ref"); err != nil {
|
||||
return Context{}, err
|
||||
}
|
||||
positions := make(map[int]int, len(value.Units))
|
||||
for unitIndex := range value.Units {
|
||||
unit := value.Units[unitIndex]
|
||||
seenUnitIDs := make(map[int]struct{}, len(value))
|
||||
sourceID := ""
|
||||
for unitIndex := range value {
|
||||
unit := value[unitIndex]
|
||||
if unit.ID <= 0 || strings.TrimSpace(unit.Kind) == "" || strings.TrimSpace(unit.Text) == "" {
|
||||
return Context{}, fmt.Errorf("%s.units[%d] has invalid required fields", prefix, unitIndex)
|
||||
return nil, fmt.Errorf("units[%d] has invalid required fields", unitIndex)
|
||||
}
|
||||
if err := validateRefIdentity(sourceID, unit.Ref, fmt.Sprintf("%s.units[%d].ref", prefix, unitIndex)); err != nil {
|
||||
return Context{}, err
|
||||
if err := validateUnitRef(unit, unitIndex); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if unit.Ref.StartUnitID != unit.ID || unit.Ref.EndUnitID != unit.ID {
|
||||
return Context{}, fmt.Errorf("%s.units[%d].ref must identify unit id %d", prefix, unitIndex, unit.ID)
|
||||
if sourceID == "" {
|
||||
sourceID = unit.Ref.SourceID
|
||||
} else if unit.Ref.SourceID != sourceID {
|
||||
return nil, fmt.Errorf("units[%d].ref.source_id must match units[0].ref.source_id", unitIndex)
|
||||
}
|
||||
if _, exists := positions[unit.ID]; exists {
|
||||
return Context{}, fmt.Errorf("%s.units contains duplicate unit id %d", prefix, unit.ID)
|
||||
}
|
||||
if _, exists := seenUnits[unit.ID]; exists {
|
||||
return Context{}, fmt.Errorf("contexts contain duplicate unit id %d", unit.ID)
|
||||
}
|
||||
positions[unit.ID] = unitIndex
|
||||
seenUnits[unit.ID] = struct{}{}
|
||||
}
|
||||
if value.ContextRef.StartUnitID != value.Units[0].ID || value.ContextRef.EndUnitID != value.Units[len(value.Units)-1].ID {
|
||||
return Context{}, fmt.Errorf("%s.context_ref must identify the first and last units", prefix)
|
||||
}
|
||||
for evidenceIndex := range value.EvidenceRefs {
|
||||
evidence := value.EvidenceRefs[evidenceIndex]
|
||||
if _, ok := selected[evidence.LaneID]; !ok {
|
||||
return Context{}, fmt.Errorf("%s.evidence_refs[%d].lane_id is not selected", prefix, evidenceIndex)
|
||||
}
|
||||
if err := requireIdentity(fmt.Sprintf("%s.evidence_refs[%d].lane_id", prefix, evidenceIndex), evidence.LaneID); err != nil {
|
||||
return Context{}, err
|
||||
}
|
||||
if err := validateRefIdentity(sourceID, evidence.SourceRef, fmt.Sprintf("%s.evidence_refs[%d].source_ref", prefix, evidenceIndex)); err != nil {
|
||||
return Context{}, err
|
||||
}
|
||||
start, startOK := positions[evidence.SourceRef.StartUnitID]
|
||||
end, endOK := positions[evidence.SourceRef.EndUnitID]
|
||||
if !startOK || !endOK || start > end {
|
||||
return Context{}, fmt.Errorf("%s.evidence_refs[%d].source_ref is outside context units", prefix, evidenceIndex)
|
||||
}
|
||||
if evidenceIndex > 0 && !lessEvidenceRef(value.EvidenceRefs[evidenceIndex-1], evidence) {
|
||||
return Context{}, fmt.Errorf("%s.evidence_refs must be unique and in deterministic order", prefix)
|
||||
if _, exists := seenUnitIDs[unit.ID]; exists {
|
||||
return nil, fmt.Errorf("units contains duplicate unit id %d", unit.ID)
|
||||
}
|
||||
seenUnitIDs[unit.ID] = struct{}{}
|
||||
}
|
||||
return value, nil
|
||||
}
|
||||
|
||||
func validateRefIdentity(sourceID string, ref source.SourceRef, field string) error {
|
||||
if ref.SourceID != sourceID {
|
||||
return fmt.Errorf("%s.source_id does not match source_id", field)
|
||||
func validateUnitRef(unit source.SourceUnit, unitIndex int) error {
|
||||
prefix := fmt.Sprintf("units[%d].ref", unitIndex)
|
||||
if strings.TrimSpace(unit.Ref.SourceID) == "" || strings.TrimSpace(unit.Ref.SourceID) != unit.Ref.SourceID {
|
||||
return fmt.Errorf("%s.source_id must be a non-empty trimmed string", prefix)
|
||||
}
|
||||
if ref.StartUnitID <= 0 || ref.EndUnitID <= 0 {
|
||||
return fmt.Errorf("%s endpoints must be positive", field)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func requireIdentity(field, value string) error {
|
||||
if strings.TrimSpace(value) == "" || strings.TrimSpace(value) != value {
|
||||
return fmt.Errorf("%s must be a non-empty trimmed string", field)
|
||||
if unit.Ref.StartUnitID != unit.ID || unit.Ref.EndUnitID != unit.ID {
|
||||
return fmt.Errorf("%s must identify unit id %d", prefix, unit.ID)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func clone(value Document) (Document, error) {
|
||||
value.SelectedLanes = append([]string(nil), value.SelectedLanes...)
|
||||
if value.Contexts == nil {
|
||||
value.Contexts = make([]Context, 0)
|
||||
} else {
|
||||
contexts := make([]Context, len(value.Contexts))
|
||||
for contextIndex, context := range value.Contexts {
|
||||
contexts[contextIndex].ContextRef = context.ContextRef
|
||||
contexts[contextIndex].EvidenceRefs = append([]EvidenceRef(nil), context.EvidenceRefs...)
|
||||
contexts[contextIndex].Units = make([]source.SourceUnit, len(context.Units))
|
||||
for unitIndex, unit := range context.Units {
|
||||
cloned, err := cloneSourceUnit(unit)
|
||||
if err != nil {
|
||||
return Document{}, fmt.Errorf("clone contexts[%d].units[%d]: %w", contextIndex, unitIndex, err)
|
||||
}
|
||||
contexts[contextIndex].Units[unitIndex] = cloned
|
||||
}
|
||||
}
|
||||
value.Contexts = contexts
|
||||
if value == nil {
|
||||
return nil, nil
|
||||
}
|
||||
return value, nil
|
||||
cloned := make(Document, len(value))
|
||||
for unitIndex, unit := range value {
|
||||
owned, err := cloneSourceUnit(unit)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("clone units[%d]: %w", unitIndex, err)
|
||||
}
|
||||
cloned[unitIndex] = owned
|
||||
}
|
||||
return cloned, nil
|
||||
}
|
||||
|
||||
func cloneSourceUnit(unit source.SourceUnit) (source.SourceUnit, error) {
|
||||
|
||||
@@ -2,7 +2,6 @@ package evidencecontext
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"math"
|
||||
"os"
|
||||
"reflect"
|
||||
@@ -12,126 +11,57 @@ import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
)
|
||||
|
||||
func TestBuildExpandsAndMergesEvidenceByDocumentPosition(t *testing.T) {
|
||||
func TestBuildSelectsExpandedSourceUnitUnion(t *testing.T) {
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
window int
|
||||
evidence []LaneEvidence
|
||||
wantUnits [][]int
|
||||
wantRefs [][]EvidenceRef
|
||||
name string
|
||||
window int
|
||||
refs []source.SourceRef
|
||||
wantIDs []int
|
||||
}{
|
||||
{
|
||||
name: "zero window",
|
||||
evidence: []LaneEvidence{{LaneID: "npcs", SourceRefs: []source.SourceRef{ref(3, 3)}}},
|
||||
wantUnits: [][]int{{3}},
|
||||
wantRefs: [][]EvidenceRef{{{LaneID: "npcs", SourceRef: ref(3, 3)}}},
|
||||
},
|
||||
{
|
||||
name: "non monotonic ids use positions and clip boundaries",
|
||||
window: 1,
|
||||
evidence: []LaneEvidence{{LaneID: "npcs", SourceRefs: []source.SourceRef{ref(3, 3)}}},
|
||||
wantUnits: [][]int{{10, 3, 30}},
|
||||
wantRefs: [][]EvidenceRef{{{LaneID: "npcs", SourceRef: ref(3, 3)}}},
|
||||
},
|
||||
{
|
||||
name: "separate gaps stay separate",
|
||||
evidence: []LaneEvidence{{LaneID: "npcs", SourceRefs: []source.SourceRef{ref(10, 10), ref(50, 50)}}},
|
||||
wantUnits: [][]int{{10}, {50}},
|
||||
wantRefs: [][]EvidenceRef{{{LaneID: "npcs", SourceRef: ref(10, 10)}}, {{LaneID: "npcs", SourceRef: ref(50, 50)}}},
|
||||
},
|
||||
{
|
||||
name: "overlapping windows merge",
|
||||
window: 1,
|
||||
evidence: []LaneEvidence{{LaneID: "npcs", SourceRefs: []source.SourceRef{ref(3, 3), ref(30, 30)}}},
|
||||
wantUnits: [][]int{{10, 3, 30, 7}},
|
||||
wantRefs: [][]EvidenceRef{{{LaneID: "npcs", SourceRef: ref(3, 3)}, {LaneID: "npcs", SourceRef: ref(30, 30)}}},
|
||||
},
|
||||
{
|
||||
name: "contiguous windows merge",
|
||||
window: 1,
|
||||
evidence: []LaneEvidence{{LaneID: "npcs", SourceRefs: []source.SourceRef{ref(10, 10), ref(7, 7)}}},
|
||||
wantUnits: [][]int{{10, 3, 30, 7, 50}},
|
||||
wantRefs: [][]EvidenceRef{{{LaneID: "npcs", SourceRef: ref(7, 7)}, {LaneID: "npcs", SourceRef: ref(10, 10)}}},
|
||||
},
|
||||
{
|
||||
name: "duplicate contributions retain unique lane attribution",
|
||||
evidence: []LaneEvidence{
|
||||
{LaneID: "spells", SourceRefs: []source.SourceRef{ref(30, 30), ref(30, 30)}},
|
||||
{LaneID: "npcs", SourceRefs: []source.SourceRef{ref(30, 30)}},
|
||||
},
|
||||
wantUnits: [][]int{{30}},
|
||||
wantRefs: [][]EvidenceRef{{{LaneID: "npcs", SourceRef: ref(30, 30)}, {LaneID: "spells", SourceRef: ref(30, 30)}}},
|
||||
},
|
||||
{
|
||||
name: "empty contributions retain explicit empty contexts",
|
||||
evidence: []LaneEvidence{{LaneID: "npcs"}},
|
||||
wantUnits: [][]int{},
|
||||
wantRefs: [][]EvidenceRef{},
|
||||
},
|
||||
{
|
||||
name: "largest window clips without overflow",
|
||||
window: math.MaxInt,
|
||||
evidence: []LaneEvidence{{LaneID: "npcs", SourceRefs: []source.SourceRef{ref(30, 30)}}},
|
||||
wantUnits: [][]int{{10, 3, 30, 7, 50}},
|
||||
wantRefs: [][]EvidenceRef{{{LaneID: "npcs", SourceRef: ref(30, 30)}}},
|
||||
},
|
||||
{name: "zero window", refs: []source.SourceRef{ref(3, 3)}, wantIDs: []int{3}},
|
||||
{name: "multi unit citation includes complete range", refs: []source.SourceRef{ref(3, 7)}, wantIDs: []int{3, 30, 7}},
|
||||
{name: "non monotonic IDs use document positions", window: 1, refs: []source.SourceRef{ref(3, 3)}, wantIDs: []int{10, 3, 30}},
|
||||
{name: "boundary clamping", window: 1, refs: []source.SourceRef{ref(10, 10), ref(50, 50)}, wantIDs: []int{10, 3, 7, 50}},
|
||||
{name: "overlapping and adjacent windows merge", window: 1, refs: []source.SourceRef{ref(3, 3), ref(30, 30), ref(30, 30)}, wantIDs: []int{10, 3, 30, 7}},
|
||||
{name: "adjacent expanded ranges merge", window: 1, refs: []source.SourceRef{ref(10, 10), ref(7, 7)}, wantIDs: []int{10, 3, 30, 7, 50}},
|
||||
{name: "largest window clips without overflow", window: math.MaxInt, refs: []source.SourceRef{ref(30, 30)}, wantIDs: []int{10, 3, 30, 7, 50}},
|
||||
{name: "no references returns an initialized empty document", wantIDs: []int{}},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
document := testDocument(t)
|
||||
got, err := Build(BuildRequest{Source: document, WindowUnits: test.window, SelectedLanes: []string{"spells", "npcs"}, LaneEvidence: test.evidence})
|
||||
got, err := Build(BuildRequest{Source: testDocument(t), WindowUnits: test.window, SourceRefs: test.refs})
|
||||
if err != nil {
|
||||
t.Fatalf("Build() error = %v", err)
|
||||
}
|
||||
if want := []string{"npcs", "spells"}; !reflect.DeepEqual(got.SelectedLanes, want) {
|
||||
t.Fatalf("SelectedLanes = %#v, want %#v", got.SelectedLanes, want)
|
||||
if got == nil {
|
||||
t.Fatal("Build() returned a nil document")
|
||||
}
|
||||
if got.WindowUnits != test.window || got.SourceID != document.ID || got.SourceDigest != document.Digest {
|
||||
t.Fatalf("Build() identity = %#v, want source and window identity", got)
|
||||
}
|
||||
if actual := contextUnitIDs(got.Contexts); !reflect.DeepEqual(actual, test.wantUnits) {
|
||||
t.Fatalf("context unit ids = %#v, want %#v", actual, test.wantUnits)
|
||||
}
|
||||
if actual := contextEvidenceRefs(got.Contexts); !reflect.DeepEqual(actual, test.wantRefs) {
|
||||
t.Fatalf("context evidence refs = %#v, want %#v", actual, test.wantRefs)
|
||||
if actual := unitIDs(got); !reflect.DeepEqual(actual, test.wantIDs) {
|
||||
t.Fatalf("unit IDs = %#v, want %#v", actual, test.wantIDs)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildIsStableAndOwnsSourceAndInputs(t *testing.T) {
|
||||
func TestBuildCopiesSelectedUnitsAndMetadata(t *testing.T) {
|
||||
document := testDocument(t)
|
||||
refs := []source.SourceRef{ref(30, 30), ref(3, 3)}
|
||||
request := BuildRequest{
|
||||
Source: document,
|
||||
WindowUnits: 1,
|
||||
SelectedLanes: []string{"spells", "npcs"},
|
||||
LaneEvidence: []LaneEvidence{{LaneID: "spells", SourceRefs: refs}, {LaneID: "npcs", SourceRefs: []source.SourceRef{ref(3, 3)}}},
|
||||
}
|
||||
first, err := Build(request)
|
||||
first, err := Build(BuildRequest{Source: document, WindowUnits: 1, SourceRefs: []source.SourceRef{ref(3, 3)}})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
secondRequest := request
|
||||
secondRequest.LaneEvidence = []LaneEvidence{{LaneID: "npcs", SourceRefs: []source.SourceRef{ref(3, 3)}}, {LaneID: "spells", SourceRefs: []source.SourceRef{ref(3, 3), ref(30, 30)}}}
|
||||
second, err := Build(secondRequest)
|
||||
second, err := Build(BuildRequest{Source: document, WindowUnits: 1, SourceRefs: []source.SourceRef{ref(3, 3)}})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !reflect.DeepEqual(first, second) {
|
||||
t.Fatalf("Build() order differs:\nfirst: %#v\nsecond: %#v", first, second)
|
||||
if !reflect.DeepEqual(first[0], document.Units[0]) {
|
||||
t.Fatalf("first unit = %#v, want unchanged source unit %#v", first[0], document.Units[0])
|
||||
}
|
||||
first.SelectedLanes[0] = "changed"
|
||||
first.Contexts[0].Units[0].Metadata["nested"].(map[string]any)["value"] = "changed"
|
||||
first[0].Metadata["nested"].(map[string]any)["value"] = "changed"
|
||||
if document.Units[0].Metadata["nested"].(map[string]any)["value"] != "original" {
|
||||
t.Fatal("Build() returned metadata aliases to source document")
|
||||
t.Fatal("Build() returned metadata aliases to the source document")
|
||||
}
|
||||
document.Units[0].Metadata["nested"].(map[string]any)["value"] = "later"
|
||||
if second.Contexts[0].Units[0].Metadata["nested"].(map[string]any)["value"] != "original" {
|
||||
t.Fatal("Build() retained metadata aliases to source document")
|
||||
}
|
||||
refs[0].StartUnitID = 999
|
||||
if !containsEvidenceRef(second.Contexts[0].EvidenceRefs, ref(30, 30)) {
|
||||
t.Fatal("Build() retained source-reference input aliases")
|
||||
if second[0].Metadata["nested"].(map[string]any)["value"] != "original" {
|
||||
t.Fatal("Build() retained metadata aliases to the source document")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -142,18 +72,11 @@ func TestBuildRejectsInvalidInputs(t *testing.T) {
|
||||
want string
|
||||
}{
|
||||
{name: "negative window", mutate: func(request *BuildRequest) { request.WindowUnits = -1 }, want: "window_units"},
|
||||
{name: "blank selected lane", mutate: func(request *BuildRequest) { request.SelectedLanes = []string{" "} }, want: "selected_lanes"},
|
||||
{name: "duplicate selected lane", mutate: func(request *BuildRequest) { request.SelectedLanes = []string{"npcs", " npcs "} }, want: "duplicated"},
|
||||
{name: "unselected contribution", mutate: func(request *BuildRequest) {
|
||||
request.LaneEvidence = []LaneEvidence{{LaneID: "other", SourceRefs: []source.SourceRef{ref(3, 3)}}}
|
||||
}, want: "not selected"},
|
||||
{name: "source digest mismatch", mutate: func(request *BuildRequest) { request.Source.Digest = "sha256:" + strings.Repeat("0", 64) }, want: "does not match"},
|
||||
{name: "invalid reference", mutate: func(request *BuildRequest) {
|
||||
request.LaneEvidence = []LaneEvidence{{LaneID: "npcs", SourceRefs: []source.SourceRef{ref(99, 99)}}}
|
||||
}, want: "source reference[0]"},
|
||||
{name: "invalid reference", mutate: func(request *BuildRequest) { request.SourceRefs = []source.SourceRef{ref(99, 99)} }, want: "source reference[0]"},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
request := BuildRequest{Source: testDocument(t), SelectedLanes: []string{"npcs"}, LaneEvidence: []LaneEvidence{{LaneID: "npcs", SourceRefs: []source.SourceRef{ref(3, 3)}}}}
|
||||
request := BuildRequest{Source: testDocument(t), SourceRefs: []source.SourceRef{ref(3, 3)}}
|
||||
test.mutate(&request)
|
||||
if _, err := Build(request); err == nil || !strings.Contains(err.Error(), test.want) {
|
||||
t.Fatalf("Build() error = %v, want %q", err, test.want)
|
||||
@@ -162,7 +85,7 @@ func TestBuildRejectsInvalidInputs(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestCodecRoundTripsCompactFixtureAndOwnsDecodedValues(t *testing.T) {
|
||||
func TestCodecRoundTripsFixtureAndOwnsValues(t *testing.T) {
|
||||
fixture, err := os.ReadFile("testdata/source_evidence_context.v1.json")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
@@ -179,15 +102,16 @@ func TestCodecRoundTripsCompactFixtureAndOwnsDecodedValues(t *testing.T) {
|
||||
if !bytes.Equal(encoded, bytes.TrimSpace(fixture)) {
|
||||
t.Fatalf("fixture does not use canonical encoding\nwant: %s\n got: %s", fixture, encoded)
|
||||
}
|
||||
value.Contexts[0].Units[0].Text = "changed"
|
||||
value[0].Text = "changed"
|
||||
decoded, err := codec.Decode(encoded)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if decoded.Contexts[0].Units[0].Text != "The party meets Rowan." {
|
||||
if decoded[0].Text != "The party meets Rowan." {
|
||||
t.Fatal("Encode() retained mutable document storage")
|
||||
}
|
||||
built, err := Build(BuildRequest{Source: testDocument(t), WindowUnits: 1, SelectedLanes: []string{"npcs"}, LaneEvidence: []LaneEvidence{{LaneID: "npcs", SourceRefs: []source.SourceRef{ref(3, 3)}}}})
|
||||
|
||||
built, err := Build(BuildRequest{Source: testDocument(t), WindowUnits: 1, SourceRefs: []source.SourceRef{ref(3, 3)}})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -203,82 +127,52 @@ func TestCodecRoundTripsCompactFixtureAndOwnsDecodedValues(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
first.Contexts[0].Units[0].Metadata["nested"].(map[string]any)["value"] = "changed"
|
||||
if second.Contexts[0].Units[0].Metadata["nested"].(map[string]any)["value"] != "original" {
|
||||
first[0].Metadata["nested"].(map[string]any)["value"] = "changed"
|
||||
if second[0].Metadata["nested"].(map[string]any)["value"] != "original" {
|
||||
t.Fatal("Decode() returned metadata aliases")
|
||||
}
|
||||
}
|
||||
|
||||
func TestCodecRejectsInvalidDurableBoundaries(t *testing.T) {
|
||||
value, err := Build(BuildRequest{Source: testDocument(t), SelectedLanes: []string{"npcs"}, LaneEvidence: []LaneEvidence{{LaneID: "npcs", SourceRefs: []source.SourceRef{ref(3, 3)}}}})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
func TestCodecRejectsInvalidDurablePayloads(t *testing.T) {
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
mutate func(*Document)
|
||||
name string
|
||||
content string
|
||||
}{
|
||||
{name: "unsorted lanes", mutate: func(value *Document) { value.SelectedLanes = []string{"z", "a"} }},
|
||||
{name: "context range mismatch", mutate: func(value *Document) { value.Contexts[0].ContextRef.EndUnitID = 999 }},
|
||||
{name: "mismatched evidence source", mutate: func(value *Document) { value.Contexts[0].EvidenceRefs[0].SourceRef.SourceID = "other" }},
|
||||
{name: "invalid evidence range", mutate: func(value *Document) {
|
||||
value.Contexts[0].EvidenceRefs[0].SourceRef.StartUnitID = 10
|
||||
}},
|
||||
{name: "duplicate context unit", mutate: func(value *Document) { value.Contexts = append(value.Contexts, value.Contexts[0]) }},
|
||||
{name: "null", content: "null"},
|
||||
{name: "wrapper object", content: `{"units":[]}`},
|
||||
{name: "missing required unit field", content: `[{"id":1,"kind":"segment","ref":{"source_id":"session","start_unit_id":1,"end_unit_id":1}}]`},
|
||||
{name: "unknown unit field", content: `[{"id":1,"kind":"segment","text":"text","ref":{"source_id":"session","start_unit_id":1,"end_unit_id":1},"unknown":true}]`},
|
||||
{name: "unknown reference field", content: `[{"id":1,"kind":"segment","text":"text","ref":{"source_id":"session","start_unit_id":1,"end_unit_id":1,"unknown":true}}]`},
|
||||
{name: "invalid self reference", content: `[{"id":1,"kind":"segment","text":"text","ref":{"source_id":"session","start_unit_id":1,"end_unit_id":2}}]`},
|
||||
{name: "mixed source documents", content: `[{"id":1,"kind":"segment","text":"one","ref":{"source_id":"session-one","start_unit_id":1,"end_unit_id":1}},{"id":2,"kind":"segment","text":"two","ref":{"source_id":"session-two","start_unit_id":2,"end_unit_id":2}}]`},
|
||||
{name: "duplicate units", content: `[{"id":1,"kind":"segment","text":"one","ref":{"source_id":"session","start_unit_id":1,"end_unit_id":1}},{"id":1,"kind":"segment","text":"two","ref":{"source_id":"session","start_unit_id":1,"end_unit_id":1}}]`},
|
||||
{name: "multiple JSON values", content: `[] []`},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
candidate, err := clone(value)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
test.mutate(&candidate)
|
||||
if _, err := New().Encode(candidate); err == nil {
|
||||
t.Fatal("Encode() error = nil, want durable model rejection")
|
||||
}
|
||||
})
|
||||
}
|
||||
content, err := New().Encode(value)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
mutate func(map[string]any)
|
||||
}{
|
||||
{name: "missing contexts", mutate: func(value map[string]any) { delete(value, "contexts") }},
|
||||
{name: "null contexts", mutate: func(value map[string]any) { value["contexts"] = nil }},
|
||||
{name: "unknown fixed field", mutate: func(value map[string]any) { value["unknown"] = true }},
|
||||
{name: "missing units", mutate: func(value map[string]any) { delete(contextObject(value, 0), "units") }},
|
||||
{name: "null evidence refs", mutate: func(value map[string]any) { contextObject(value, 0)["evidence_refs"] = nil }},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
raw := decodeJSON(t, content)
|
||||
test.mutate(raw)
|
||||
mutated, err := json.Marshal(raw)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := New().Decode(mutated); err == nil {
|
||||
if _, err := New().Decode([]byte(test.content)); err == nil {
|
||||
t.Fatal("Decode() error = nil, want strict payload rejection")
|
||||
}
|
||||
})
|
||||
}
|
||||
if _, err := New().Decode(append(content, []byte(" {}")...)); err == nil {
|
||||
t.Fatal("Decode() error = nil, want trailing JSON rejection")
|
||||
if _, err := New().Encode(nil); err == nil {
|
||||
t.Fatal("Encode(nil) error = nil, want array rejection")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSerializeUsesFixedArtifactIdentity(t *testing.T) {
|
||||
artifact, err := Serialize(BuildRequest{Source: testDocument(t), SelectedLanes: []string{"npcs"}})
|
||||
func TestSerializeUsesFixedArtifactIdentityAndEmptyArray(t *testing.T) {
|
||||
artifact, err := Serialize(BuildRequest{Source: testDocument(t)})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if artifact.Kind != ArtifactKind || artifact.MediaType != MediaType || artifact.Schema.ID != SchemaID || artifact.Schema.Name != SchemaName || artifact.Schema.Version != SchemaVersion {
|
||||
t.Fatalf("Serialize() = %#v, want fixed artifact identity", artifact)
|
||||
}
|
||||
if string(artifact.Content) != "[]" {
|
||||
t.Fatalf("Serialize() content = %s, want []", artifact.Content)
|
||||
}
|
||||
decoded, err := New().Decode(artifact.Content)
|
||||
if err != nil || len(decoded.Contexts) != 0 || decoded.Contexts == nil {
|
||||
t.Fatalf("Decode(Serialize()) = %#v, %v; want explicit empty contexts", decoded, err)
|
||||
if err != nil || decoded == nil || len(decoded) != 0 {
|
||||
t.Fatalf("Decode(Serialize()) = %#v, %v; want explicit empty array", decoded, err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -306,45 +200,10 @@ func ref(start, end int) source.SourceRef {
|
||||
return source.SourceRef{SourceID: "session", StartUnitID: start, EndUnitID: end}
|
||||
}
|
||||
|
||||
func contextUnitIDs(contexts []Context) [][]int {
|
||||
values := make([][]int, len(contexts))
|
||||
for index, context := range contexts {
|
||||
values[index] = make([]int, len(context.Units))
|
||||
for unitIndex, unit := range context.Units {
|
||||
values[index][unitIndex] = unit.ID
|
||||
}
|
||||
func unitIDs(units Document) []int {
|
||||
values := make([]int, len(units))
|
||||
for index, unit := range units {
|
||||
values[index] = unit.ID
|
||||
}
|
||||
return values
|
||||
}
|
||||
|
||||
func contextEvidenceRefs(contexts []Context) [][]EvidenceRef {
|
||||
values := make([][]EvidenceRef, len(contexts))
|
||||
for index, context := range contexts {
|
||||
values[index] = append([]EvidenceRef(nil), context.EvidenceRefs...)
|
||||
}
|
||||
return values
|
||||
}
|
||||
|
||||
func containsEvidenceRef(values []EvidenceRef, want source.SourceRef) bool {
|
||||
for _, value := range values {
|
||||
if value.SourceRef == want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func decodeJSON(t *testing.T, content []byte) map[string]any {
|
||||
t.Helper()
|
||||
decoder := json.NewDecoder(bytes.NewReader(content))
|
||||
decoder.UseNumber()
|
||||
var value map[string]any
|
||||
if err := decoder.Decode(&value); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return value
|
||||
}
|
||||
|
||||
func contextObject(value map[string]any, index int) map[string]any {
|
||||
return value["contexts"].([]any)[index].(map[string]any)
|
||||
}
|
||||
|
||||
@@ -14,37 +14,12 @@ const (
|
||||
MediaType = "application/json"
|
||||
)
|
||||
|
||||
// Document is the durable union of direct evidence and surrounding source
|
||||
// context selected for one accepted source document.
|
||||
type Document struct {
|
||||
SourceID string `json:"source_id"`
|
||||
SourceDigest string `json:"source_digest"`
|
||||
WindowUnits int `json:"window_units"`
|
||||
SelectedLanes []string `json:"selected_lanes"`
|
||||
Contexts []Context `json:"contexts"`
|
||||
}
|
||||
// Document is the durable selected source-unit excerpt.
|
||||
type Document []source.SourceUnit
|
||||
|
||||
type Context struct {
|
||||
ContextRef source.SourceRef `json:"context_ref"`
|
||||
EvidenceRefs []EvidenceRef `json:"evidence_refs"`
|
||||
Units []source.SourceUnit `json:"units"`
|
||||
}
|
||||
|
||||
type EvidenceRef struct {
|
||||
LaneID string `json:"lane_id"`
|
||||
SourceRef source.SourceRef `json:"source_ref"`
|
||||
}
|
||||
|
||||
// LaneEvidence attributes direct source references to one selected lane.
|
||||
type LaneEvidence struct {
|
||||
LaneID string `json:"lane_id"`
|
||||
SourceRefs []source.SourceRef `json:"source_refs"`
|
||||
}
|
||||
|
||||
// BuildRequest supplies accepted source material and direct lane evidence.
|
||||
// BuildRequest supplies accepted source material and projected source references.
|
||||
type BuildRequest struct {
|
||||
Source *source.SourceDocument
|
||||
WindowUnits int
|
||||
SelectedLanes []string
|
||||
LaneEvidence []LaneEvidence
|
||||
Source *source.SourceDocument
|
||||
WindowUnits int
|
||||
SourceRefs []source.SourceRef
|
||||
}
|
||||
|
||||
@@ -1 +1 @@
|
||||
{"source_id":"session-alpha","source_digest":"sha256:0000000000000000000000000000000000000000000000000000000000000000","window_units":0,"selected_lanes":["npcs"],"contexts":[{"context_ref":{"source_id":"session-alpha","start_unit_id":13,"end_unit_id":13},"evidence_refs":[{"lane_id":"npcs","source_ref":{"source_id":"session-alpha","start_unit_id":13,"end_unit_id":13}}],"units":[{"id":13,"kind":"transcript_segment","text":"The party meets Rowan.","ref":{"source_id":"session-alpha","start_unit_id":13,"end_unit_id":13}}]}]}
|
||||
[{"id":13,"kind":"transcript_segment","text":"The party meets Rowan.","ref":{"source_id":"session-alpha","start_unit_id":13,"end_unit_id":13}}]
|
||||
|
||||
@@ -20,7 +20,6 @@ type debugEvidenceContextSummary struct {
|
||||
SchemaVersion string `json:"schema_version"`
|
||||
SelectedLanes []string `json:"selected_lanes"`
|
||||
WindowUnits int `json:"window_units"`
|
||||
ContextCount int `json:"context_count"`
|
||||
UnitCount int `json:"unit_count"`
|
||||
SourceDigest string `json:"source_digest"`
|
||||
}
|
||||
@@ -49,10 +48,9 @@ func buildOutputEvidenceContext(prepared *PreparedPipeline, doc *source.SourceDo
|
||||
}
|
||||
|
||||
request := evidencecontext.BuildRequest{
|
||||
Source: doc,
|
||||
WindowUnits: prepared.evidencePlan.policy.WindowUnits,
|
||||
SelectedLanes: append([]string(nil), prepared.evidencePlan.policy.LaneIDs...),
|
||||
LaneEvidence: make([]evidencecontext.LaneEvidence, 0, len(prepared.evidencePlan.lanes)),
|
||||
Source: doc,
|
||||
WindowUnits: prepared.evidencePlan.policy.WindowUnits,
|
||||
SourceRefs: make([]source.SourceRef, 0),
|
||||
}
|
||||
for _, lane := range prepared.evidencePlan.lanes {
|
||||
output, ok := byLane[lane.laneID]
|
||||
@@ -73,10 +71,7 @@ func buildOutputEvidenceContext(prepared *PreparedPipeline, doc *source.SourceDo
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("evidence context output lane %q: accepted normalized artifact cannot be projected", lane.laneID)
|
||||
}
|
||||
request.LaneEvidence = append(request.LaneEvidence, evidencecontext.LaneEvidence{
|
||||
LaneID: lane.laneID,
|
||||
SourceRefs: append([]source.SourceRef(nil), references...),
|
||||
})
|
||||
request.SourceRefs = append(request.SourceRefs, references...)
|
||||
}
|
||||
|
||||
document, err := evidencecontext.Build(request)
|
||||
@@ -99,13 +94,10 @@ func buildOutputEvidenceContext(prepared *PreparedPipeline, doc *source.SourceDo
|
||||
SchemaID: artifact.Schema.ID,
|
||||
SchemaName: artifact.Schema.Name,
|
||||
SchemaVersion: artifact.Schema.Version,
|
||||
SelectedLanes: append([]string(nil), document.SelectedLanes...),
|
||||
WindowUnits: document.WindowUnits,
|
||||
ContextCount: len(document.Contexts),
|
||||
SourceDigest: document.SourceDigest,
|
||||
}
|
||||
for _, context := range document.Contexts {
|
||||
summary.UnitCount += len(context.Units)
|
||||
SelectedLanes: append([]string(nil), prepared.evidencePlan.policy.LaneIDs...),
|
||||
WindowUnits: prepared.evidencePlan.policy.WindowUnits,
|
||||
SourceDigest: doc.Digest,
|
||||
UnitCount: len(document),
|
||||
}
|
||||
return contracts.CloneSerializedArtifactPointer(artifact), &summary, nil
|
||||
}
|
||||
|
||||
@@ -97,14 +97,11 @@ func TestRunnerBuildsEvidenceContextFromSelectedNormalizedOutputs(t *testing.T)
|
||||
}
|
||||
|
||||
value := decodeCapturedEvidence(t, encoder)
|
||||
if !reflect.DeepEqual(value.SelectedLanes, []string{"alpha", "beta", "inactive"}) || len(value.Contexts) != 1 || len(value.Contexts[0].Units) != 3 {
|
||||
if actual := []int{value[0].ID, value[1].ID, value[2].ID}; !reflect.DeepEqual(actual, []int{1, 2, 3}) {
|
||||
t.Fatalf("evidence context = %#v, want selected union", value)
|
||||
}
|
||||
if got := value.Contexts[0].EvidenceRefs; len(got) != 2 || got[0].LaneID != "alpha" || got[1].LaneID != "beta" {
|
||||
t.Fatalf("evidence refs = %#v, want both selected lanes", got)
|
||||
}
|
||||
debugJSON := string(debug.json["output/evidence-context.json"])
|
||||
if strings.Contains(debugJSON, "text-1") || strings.Contains(debugJSON, "metadata") || !strings.Contains(debugJSON, `"artifact_kind":"source/evidence-context"`) || !strings.Contains(debugJSON, `"schema_id":"notarius.source.evidence_context"`) || !strings.Contains(debugJSON, `"context_count":1`) || !strings.Contains(debugJSON, `"unit_count":3`) {
|
||||
if strings.Contains(debugJSON, "text-1") || strings.Contains(debugJSON, "metadata") || !strings.Contains(debugJSON, `"artifact_kind":"source/evidence-context"`) || !strings.Contains(debugJSON, `"schema_id":"notarius.source.evidence_context"`) || strings.Contains(debugJSON, "context_count") || !strings.Contains(debugJSON, `"unit_count":3`) {
|
||||
t.Fatalf("evidence debug envelope = %s, want only allowlisted summary", debugJSON)
|
||||
}
|
||||
}
|
||||
@@ -134,7 +131,7 @@ func TestRunnerEvidenceContextOmitsAbsentAndRejectedLanes(t *testing.T) {
|
||||
t.Fatalf("rejections = %#v, want rejected lane unchanged", result.Rejected)
|
||||
}
|
||||
value := decodeCapturedEvidence(t, encoder)
|
||||
if len(value.Contexts) != 1 || len(value.Contexts[0].EvidenceRefs) != 1 || value.Contexts[0].EvidenceRefs[0].LaneID != "present" {
|
||||
if len(value) != 1 || value[0].ID != 1 {
|
||||
t.Fatalf("evidence context = %#v, want present lane only", value)
|
||||
}
|
||||
}
|
||||
@@ -232,7 +229,7 @@ func TestRunnerEvidenceContextRebuildsFromAcceptedCheckpoint(t *testing.T) {
|
||||
t.Fatalf("Run() error = %v", err)
|
||||
}
|
||||
value := decodeCapturedEvidence(t, encoder)
|
||||
if len(value.Contexts) != 1 || value.Contexts[0].EvidenceRefs[0].LaneID != "notes" {
|
||||
if len(value) != 1 || value[0].ID != 1 {
|
||||
t.Fatalf("evidence context = %#v, want checkpointed normalized output", value)
|
||||
}
|
||||
}
|
||||
|
||||
394
internal/framework/semanticreconcile/application.go
Normal file
394
internal/framework/semanticreconcile/application.go
Normal file
@@ -0,0 +1,394 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// CloneValueFunc returns a value that shares no caller-owned mutable state with
|
||||
// its input.
|
||||
type CloneValueFunc[T any] func(T) T
|
||||
|
||||
// Record owns one typed value and its deterministic input provenance.
|
||||
type Record[T any] struct {
|
||||
value T
|
||||
originalInputIndexes []int
|
||||
earliestInputPosition int
|
||||
cloneValue CloneValueFunc[T]
|
||||
}
|
||||
|
||||
// NewRecord constructs an owned typed record. Original input indexes are
|
||||
// normalized into ascending unique order.
|
||||
func NewRecord[T any](value T, originalInputIndexes []int, earliestInputPosition int, cloneValue CloneValueFunc[T]) (Record[T], error) {
|
||||
if cloneValue == nil {
|
||||
return Record[T]{}, fmt.Errorf("construct semantic reconciliation record: clone value function must not be nil")
|
||||
}
|
||||
if earliestInputPosition < 0 {
|
||||
return Record[T]{}, fmt.Errorf("construct semantic reconciliation record: earliest input position must not be negative")
|
||||
}
|
||||
indexes, err := normalizeInputIndexes(originalInputIndexes)
|
||||
if err != nil {
|
||||
return Record[T]{}, fmt.Errorf("construct semantic reconciliation record: %w", err)
|
||||
}
|
||||
return Record[T]{
|
||||
value: cloneValue(value),
|
||||
originalInputIndexes: indexes,
|
||||
earliestInputPosition: earliestInputPosition,
|
||||
cloneValue: cloneValue,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// Value returns an independently owned typed value.
|
||||
func (record Record[T]) Value() T {
|
||||
if record.cloneValue == nil {
|
||||
var zero T
|
||||
return zero
|
||||
}
|
||||
return record.cloneValue(record.value)
|
||||
}
|
||||
|
||||
// OriginalInputIndexes returns an owned ascending unique index list.
|
||||
func (record Record[T]) OriginalInputIndexes() []int {
|
||||
return cloneSlice(record.originalInputIndexes)
|
||||
}
|
||||
|
||||
// EarliestInputPosition returns the earliest deterministic input position
|
||||
// contributing to this record.
|
||||
func (record Record[T]) EarliestInputPosition() int { return record.earliestInputPosition }
|
||||
|
||||
// RejectionCategory is a stable, domain-neutral reason that an otherwise safe
|
||||
// semantic group was not applied.
|
||||
type RejectionCategory string
|
||||
|
||||
// ApplicationPolicy supplies only the typed behavior needed to apply a safe
|
||||
// reconciliation plan. An empty category from RejectGroup accepts the group.
|
||||
type ApplicationPolicy[T any] struct {
|
||||
CloneValue CloneValueFunc[T]
|
||||
RejectGroup func(members []T, canonical T) RejectionCategory
|
||||
ConsolidateGroup func(members []T, canonical T) (T, error)
|
||||
}
|
||||
|
||||
// GroupProvenance identifies the complete input contribution of one plan
|
||||
// group without prescribing domain warning or retry policy.
|
||||
type GroupProvenance struct {
|
||||
memberPositions []int
|
||||
canonicalPosition int
|
||||
originalInputIndexes []int
|
||||
earliestPosition int
|
||||
}
|
||||
|
||||
// MemberPositions returns owned record positions in ascending order.
|
||||
func (provenance GroupProvenance) MemberPositions() []int {
|
||||
return cloneSlice(provenance.memberPositions)
|
||||
}
|
||||
|
||||
// CanonicalPosition returns the plan-selected canonical record position.
|
||||
func (provenance GroupProvenance) CanonicalPosition() int {
|
||||
return provenance.canonicalPosition
|
||||
}
|
||||
|
||||
// OriginalInputIndexes returns the sorted union contributed by all members.
|
||||
func (provenance GroupProvenance) OriginalInputIndexes() []int {
|
||||
return cloneSlice(provenance.originalInputIndexes)
|
||||
}
|
||||
|
||||
// EarliestInputPosition returns the earliest member input position.
|
||||
func (provenance GroupProvenance) EarliestInputPosition() int {
|
||||
return provenance.earliestPosition
|
||||
}
|
||||
|
||||
// AppliedGroup records the provenance of one successfully consolidated group.
|
||||
type AppliedGroup struct {
|
||||
provenance GroupProvenance
|
||||
}
|
||||
|
||||
// Provenance returns an independently owned provenance snapshot.
|
||||
func (event AppliedGroup) Provenance() GroupProvenance {
|
||||
return cloneGroupProvenance(event.provenance)
|
||||
}
|
||||
|
||||
// RejectedGroup records a typed guard decision while leaving warning and retry
|
||||
// construction to the consuming domain.
|
||||
type RejectedGroup struct {
|
||||
category RejectionCategory
|
||||
provenance GroupProvenance
|
||||
}
|
||||
|
||||
// Category returns the stable neutral rejection category.
|
||||
func (event RejectedGroup) Category() RejectionCategory { return event.category }
|
||||
|
||||
// Provenance returns an independently owned provenance snapshot.
|
||||
func (event RejectedGroup) Provenance() GroupProvenance {
|
||||
return cloneGroupProvenance(event.provenance)
|
||||
}
|
||||
|
||||
// ApplicationResult owns the ordered records and neutral events from one plan
|
||||
// application.
|
||||
type ApplicationResult[T any] struct {
|
||||
records []Record[T]
|
||||
appliedGroups []AppliedGroup
|
||||
rejectedGroups []RejectedGroup
|
||||
}
|
||||
|
||||
// Records returns independently owned records in earliest-contribution order.
|
||||
func (result ApplicationResult[T]) Records() []Record[T] {
|
||||
return cloneRecords(result.records)
|
||||
}
|
||||
|
||||
// AppliedGroups returns independently owned applied-group events.
|
||||
func (result ApplicationResult[T]) AppliedGroups() []AppliedGroup {
|
||||
events := make([]AppliedGroup, len(result.appliedGroups))
|
||||
for index, event := range result.appliedGroups {
|
||||
events[index] = AppliedGroup{provenance: cloneGroupProvenance(event.provenance)}
|
||||
}
|
||||
return preserveEmptySlice(result.appliedGroups, events)
|
||||
}
|
||||
|
||||
// RejectedGroups returns independently owned rejected-group events.
|
||||
func (result ApplicationResult[T]) RejectedGroups() []RejectedGroup {
|
||||
events := make([]RejectedGroup, len(result.rejectedGroups))
|
||||
for index, event := range result.rejectedGroups {
|
||||
events[index] = RejectedGroup{category: event.category, provenance: cloneGroupProvenance(event.provenance)}
|
||||
}
|
||||
return preserveEmptySlice(result.rejectedGroups, events)
|
||||
}
|
||||
|
||||
type applicationEntry[T any] struct {
|
||||
record Record[T]
|
||||
order int
|
||||
}
|
||||
|
||||
// ApplyPlan applies safe, non-overlapping groups without mutating the plan,
|
||||
// records, or values supplied to policy callbacks.
|
||||
func ApplyPlan[T any](plan Plan, records []Record[T], policy ApplicationPolicy[T]) (ApplicationResult[T], error) {
|
||||
if err := validateApplicationPolicy(policy); err != nil {
|
||||
return ApplicationResult[T]{}, err
|
||||
}
|
||||
for index, record := range records {
|
||||
if err := validateRecord(record); err != nil {
|
||||
return ApplicationResult[T]{}, fmt.Errorf("apply semantic reconciliation plan: record %d: %w", index, err)
|
||||
}
|
||||
}
|
||||
|
||||
groups := plan.Groups()
|
||||
groupByFirstMember, err := validateApplicationPlan(groups, len(records))
|
||||
if err != nil {
|
||||
return ApplicationResult[T]{}, err
|
||||
}
|
||||
groupedPositions := make(map[int]struct{}, len(records))
|
||||
for _, group := range groups {
|
||||
for _, position := range group.memberPositions {
|
||||
groupedPositions[position] = struct{}{}
|
||||
}
|
||||
}
|
||||
|
||||
entries := make([]applicationEntry[T], 0, len(records))
|
||||
result := ApplicationResult[T]{}
|
||||
for position, record := range records {
|
||||
group, firstMember := groupByFirstMember[position]
|
||||
if !firstMember {
|
||||
if _, grouped := groupedPositions[position]; grouped {
|
||||
continue
|
||||
}
|
||||
entries = append(entries, applicationEntry[T]{record: cloneRecordWith(record, policy.CloneValue), order: position})
|
||||
continue
|
||||
}
|
||||
|
||||
provenance := groupProvenance(group, records)
|
||||
guardMembers, guardCanonical := policyInputs(group, records, policy.CloneValue)
|
||||
category := RejectionCategory("")
|
||||
if policy.RejectGroup != nil {
|
||||
category = policy.RejectGroup(guardMembers, guardCanonical)
|
||||
}
|
||||
if category != "" && strings.TrimSpace(string(category)) == "" {
|
||||
return ApplicationResult[T]{}, fmt.Errorf("apply semantic reconciliation group beginning at position %d: rejection category must not be blank", position)
|
||||
}
|
||||
if category != "" {
|
||||
for _, memberPosition := range group.memberPositions {
|
||||
entries = append(entries, applicationEntry[T]{record: cloneRecordWith(records[memberPosition], policy.CloneValue), order: memberPosition})
|
||||
}
|
||||
result.rejectedGroups = append(result.rejectedGroups, RejectedGroup{category: category, provenance: provenance})
|
||||
continue
|
||||
}
|
||||
|
||||
members, canonical := policyInputs(group, records, policy.CloneValue)
|
||||
consolidated, err := policy.ConsolidateGroup(members, canonical)
|
||||
if err != nil {
|
||||
return ApplicationResult[T]{}, fmt.Errorf("apply semantic reconciliation group beginning at position %d: consolidate: %w", position, err)
|
||||
}
|
||||
owned, err := NewRecord(consolidated, provenance.originalInputIndexes, provenance.earliestPosition, policy.CloneValue)
|
||||
if err != nil {
|
||||
return ApplicationResult[T]{}, fmt.Errorf("apply semantic reconciliation group beginning at position %d: own consolidated record: %w", position, err)
|
||||
}
|
||||
entries = append(entries, applicationEntry[T]{record: owned, order: position})
|
||||
result.appliedGroups = append(result.appliedGroups, AppliedGroup{provenance: provenance})
|
||||
}
|
||||
|
||||
sort.SliceStable(entries, func(left, right int) bool {
|
||||
if entries[left].record.earliestInputPosition == entries[right].record.earliestInputPosition {
|
||||
return entries[left].order < entries[right].order
|
||||
}
|
||||
return entries[left].record.earliestInputPosition < entries[right].record.earliestInputPosition
|
||||
})
|
||||
if records != nil {
|
||||
result.records = make([]Record[T], len(entries))
|
||||
for index, entry := range entries {
|
||||
result.records[index] = cloneRecord(entry.record)
|
||||
}
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func validateApplicationPolicy[T any](policy ApplicationPolicy[T]) error {
|
||||
if policy.CloneValue == nil {
|
||||
return fmt.Errorf("apply semantic reconciliation plan: clone value function must not be nil")
|
||||
}
|
||||
if policy.ConsolidateGroup == nil {
|
||||
return fmt.Errorf("apply semantic reconciliation plan: consolidate group function must not be nil")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func validateRecord[T any](record Record[T]) error {
|
||||
if record.cloneValue == nil {
|
||||
return fmt.Errorf("invalid construction state: clone value function must not be nil")
|
||||
}
|
||||
if record.earliestInputPosition < 0 {
|
||||
return fmt.Errorf("invalid construction state: earliest input position must not be negative")
|
||||
}
|
||||
for index, value := range record.originalInputIndexes {
|
||||
if value < 0 {
|
||||
return fmt.Errorf("invalid construction state: original input index must not be negative")
|
||||
}
|
||||
if index > 0 && record.originalInputIndexes[index-1] >= value {
|
||||
return fmt.Errorf("invalid construction state: original input indexes must be ascending and unique")
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func validateApplicationPlan(groups []PlanGroup, recordCount int) (map[int]PlanGroup, error) {
|
||||
groupByFirstMember := make(map[int]PlanGroup, len(groups))
|
||||
used := make(map[int]struct{})
|
||||
for groupIndex, group := range groups {
|
||||
if len(group.memberPositions) < 2 {
|
||||
return nil, fmt.Errorf("apply semantic reconciliation plan: group %d must contain at least two member positions", groupIndex)
|
||||
}
|
||||
canonicalMember := false
|
||||
for memberIndex, position := range group.memberPositions {
|
||||
if position < 0 || position >= recordCount {
|
||||
return nil, fmt.Errorf("apply semantic reconciliation plan: group %d member position %d is outside record range [0,%d)", groupIndex, position, recordCount)
|
||||
}
|
||||
if memberIndex > 0 && group.memberPositions[memberIndex-1] >= position {
|
||||
return nil, fmt.Errorf("apply semantic reconciliation plan: group %d member positions must be ascending and unique", groupIndex)
|
||||
}
|
||||
if _, exists := used[position]; exists {
|
||||
return nil, fmt.Errorf("apply semantic reconciliation plan: record position %d belongs to multiple groups", position)
|
||||
}
|
||||
used[position] = struct{}{}
|
||||
canonicalMember = canonicalMember || position == group.canonicalPosition
|
||||
}
|
||||
if group.canonicalPosition < 0 || group.canonicalPosition >= recordCount {
|
||||
return nil, fmt.Errorf("apply semantic reconciliation plan: group %d canonical position %d is outside record range [0,%d)", groupIndex, group.canonicalPosition, recordCount)
|
||||
}
|
||||
if !canonicalMember {
|
||||
return nil, fmt.Errorf("apply semantic reconciliation plan: group %d canonical position %d is not a member", groupIndex, group.canonicalPosition)
|
||||
}
|
||||
groupByFirstMember[group.memberPositions[0]] = group
|
||||
}
|
||||
return groupByFirstMember, nil
|
||||
}
|
||||
|
||||
func groupProvenance[T any](group PlanGroup, records []Record[T]) GroupProvenance {
|
||||
provenance := GroupProvenance{
|
||||
memberPositions: cloneSlice(group.memberPositions),
|
||||
canonicalPosition: group.canonicalPosition,
|
||||
earliestPosition: records[group.memberPositions[0]].earliestInputPosition,
|
||||
}
|
||||
for _, position := range group.memberPositions {
|
||||
provenance.originalInputIndexes = append(provenance.originalInputIndexes, records[position].originalInputIndexes...)
|
||||
if records[position].earliestInputPosition < provenance.earliestPosition {
|
||||
provenance.earliestPosition = records[position].earliestInputPosition
|
||||
}
|
||||
}
|
||||
provenance.originalInputIndexes, _ = normalizeInputIndexes(provenance.originalInputIndexes)
|
||||
return provenance
|
||||
}
|
||||
|
||||
func policyInputs[T any](group PlanGroup, records []Record[T], cloneValue CloneValueFunc[T]) ([]T, T) {
|
||||
members := make([]T, len(group.memberPositions))
|
||||
for index, position := range group.memberPositions {
|
||||
members[index] = cloneValue(records[position].value)
|
||||
}
|
||||
return members, cloneValue(records[group.canonicalPosition].value)
|
||||
}
|
||||
|
||||
func normalizeInputIndexes(indexes []int) ([]int, error) {
|
||||
if indexes == nil {
|
||||
return nil, nil
|
||||
}
|
||||
normalized := append([]int{}, indexes...)
|
||||
for _, index := range normalized {
|
||||
if index < 0 {
|
||||
return nil, fmt.Errorf("original input index must not be negative")
|
||||
}
|
||||
}
|
||||
sort.Ints(normalized)
|
||||
write := 0
|
||||
for _, index := range normalized {
|
||||
if write > 0 && normalized[write-1] == index {
|
||||
continue
|
||||
}
|
||||
normalized[write] = index
|
||||
write++
|
||||
}
|
||||
return normalized[:write], nil
|
||||
}
|
||||
|
||||
func cloneRecord[T any](record Record[T]) Record[T] {
|
||||
return cloneRecordWith(record, record.cloneValue)
|
||||
}
|
||||
|
||||
func cloneRecordWith[T any](record Record[T], cloneValue CloneValueFunc[T]) Record[T] {
|
||||
return Record[T]{
|
||||
value: cloneValue(record.value),
|
||||
originalInputIndexes: cloneSlice(record.originalInputIndexes),
|
||||
earliestInputPosition: record.earliestInputPosition,
|
||||
cloneValue: cloneValue,
|
||||
}
|
||||
}
|
||||
|
||||
func cloneRecords[T any](records []Record[T]) []Record[T] {
|
||||
if records == nil {
|
||||
return nil
|
||||
}
|
||||
cloned := make([]Record[T], len(records))
|
||||
for index, record := range records {
|
||||
cloned[index] = cloneRecord(record)
|
||||
}
|
||||
return cloned
|
||||
}
|
||||
|
||||
func cloneGroupProvenance(provenance GroupProvenance) GroupProvenance {
|
||||
return GroupProvenance{
|
||||
memberPositions: cloneSlice(provenance.memberPositions),
|
||||
canonicalPosition: provenance.canonicalPosition,
|
||||
originalInputIndexes: cloneSlice(provenance.originalInputIndexes),
|
||||
earliestPosition: provenance.earliestPosition,
|
||||
}
|
||||
}
|
||||
|
||||
func cloneSlice[T any](values []T) []T {
|
||||
if values == nil {
|
||||
return nil
|
||||
}
|
||||
return append([]T{}, values...)
|
||||
}
|
||||
|
||||
func preserveEmptySlice[S ~[]E, E any](source S, cloned []E) []E {
|
||||
if source == nil {
|
||||
return nil
|
||||
}
|
||||
return cloned
|
||||
}
|
||||
376
internal/framework/semanticreconcile/application_test.go
Normal file
376
internal/framework/semanticreconcile/application_test.go
Normal file
@@ -0,0 +1,376 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
type syntheticRecord struct {
|
||||
Name string
|
||||
Notes []string
|
||||
}
|
||||
|
||||
func cloneSyntheticRecord(record syntheticRecord) syntheticRecord {
|
||||
record.Notes = cloneSlice(record.Notes)
|
||||
return record
|
||||
}
|
||||
|
||||
func TestNewRecordOwnsValueAndNormalizesProvenance(t *testing.T) {
|
||||
value := syntheticRecord{Name: "alpha", Notes: []string{"owned"}}
|
||||
indexes := []int{3, 1, 3}
|
||||
record, err := NewRecord(value, indexes, 4, cloneSyntheticRecord)
|
||||
if err != nil {
|
||||
t.Fatalf("NewRecord() error = %v", err)
|
||||
}
|
||||
|
||||
value.Notes[0] = "caller mutation"
|
||||
indexes[0] = 99
|
||||
gotValue := record.Value()
|
||||
gotIndexes := record.OriginalInputIndexes()
|
||||
if gotValue.Name != "alpha" || !reflect.DeepEqual(gotValue.Notes, []string{"owned"}) {
|
||||
t.Fatalf("Value() = %#v, want owned original", gotValue)
|
||||
}
|
||||
if !reflect.DeepEqual(gotIndexes, []int{1, 3}) {
|
||||
t.Fatalf("OriginalInputIndexes() = %v, want [1 3]", gotIndexes)
|
||||
}
|
||||
if record.EarliestInputPosition() != 4 {
|
||||
t.Fatalf("EarliestInputPosition() = %d, want 4", record.EarliestInputPosition())
|
||||
}
|
||||
|
||||
gotValue.Notes[0] = "accessor mutation"
|
||||
gotIndexes[0] = 88
|
||||
if again := record.Value(); again.Notes[0] != "owned" {
|
||||
t.Fatalf("Value() retained accessor mutation: %#v", again)
|
||||
}
|
||||
if again := record.OriginalInputIndexes(); !reflect.DeepEqual(again, []int{1, 3}) {
|
||||
t.Fatalf("OriginalInputIndexes() retained accessor mutation: %v", again)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewRecordPreservesNilAndEmptyIndexOwnership(t *testing.T) {
|
||||
nilIndexes, err := NewRecord(syntheticRecord{}, nil, 0, cloneSyntheticRecord)
|
||||
if err != nil {
|
||||
t.Fatalf("NewRecord(nil) error = %v", err)
|
||||
}
|
||||
emptyIndexes, err := NewRecord(syntheticRecord{}, []int{}, 0, cloneSyntheticRecord)
|
||||
if err != nil {
|
||||
t.Fatalf("NewRecord(empty) error = %v", err)
|
||||
}
|
||||
if nilIndexes.OriginalInputIndexes() != nil {
|
||||
t.Fatal("nil original indexes became non-nil")
|
||||
}
|
||||
if got := emptyIndexes.OriginalInputIndexes(); got == nil || len(got) != 0 {
|
||||
t.Fatalf("empty original indexes = %#v, want non-nil empty", got)
|
||||
}
|
||||
|
||||
if _, err := NewRecord(syntheticRecord{}, nil, 0, CloneValueFunc[syntheticRecord](nil)); err == nil {
|
||||
t.Fatal("NewRecord() accepted nil clone function")
|
||||
}
|
||||
if _, err := NewRecord(syntheticRecord{}, nil, -1, cloneSyntheticRecord); err == nil {
|
||||
t.Fatal("NewRecord() accepted negative earliest position")
|
||||
}
|
||||
if _, err := NewRecord(syntheticRecord{}, []int{-1}, 0, cloneSyntheticRecord); err == nil {
|
||||
t.Fatal("NewRecord() accepted negative original input index")
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyPlanConsolidatesGroupsAndOrdersByEarliestContribution(t *testing.T) {
|
||||
records := syntheticRecords(t,
|
||||
recordFixture{"alpha", []int{4}, 4},
|
||||
recordFixture{"bravo", []int{3, 1}, 1},
|
||||
recordFixture{"charlie", []int{2}, 2},
|
||||
recordFixture{"delta", []int{2, 0}, 0},
|
||||
)
|
||||
plan := Plan{groups: []PlanGroup{
|
||||
{memberPositions: []int{0, 2}, canonicalPosition: 2},
|
||||
{memberPositions: []int{1, 3}, canonicalPosition: 1},
|
||||
}}
|
||||
var canonicalNames []string
|
||||
result, err := ApplyPlan(plan, records, ApplicationPolicy[syntheticRecord]{
|
||||
CloneValue: cloneSyntheticRecord,
|
||||
ConsolidateGroup: func(members []syntheticRecord, canonical syntheticRecord) (syntheticRecord, error) {
|
||||
canonicalNames = append(canonicalNames, canonical.Name)
|
||||
canonical.Notes = []string{members[0].Name, members[1].Name}
|
||||
return canonical, nil
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("ApplyPlan() error = %v", err)
|
||||
}
|
||||
|
||||
got := result.Records()
|
||||
if names := recordNames(got); !reflect.DeepEqual(names, []string{"bravo", "charlie"}) {
|
||||
t.Fatalf("record names = %v, want [bravo charlie]", names)
|
||||
}
|
||||
if !reflect.DeepEqual(canonicalNames, []string{"charlie", "bravo"}) {
|
||||
t.Fatalf("canonical values = %v, want [charlie bravo]", canonicalNames)
|
||||
}
|
||||
if indexes := got[0].OriginalInputIndexes(); !reflect.DeepEqual(indexes, []int{0, 1, 2, 3}) {
|
||||
t.Fatalf("first provenance indexes = %v, want [0 1 2 3]", indexes)
|
||||
}
|
||||
if indexes := got[1].OriginalInputIndexes(); !reflect.DeepEqual(indexes, []int{2, 4}) {
|
||||
t.Fatalf("second provenance indexes = %v, want [2 4]", indexes)
|
||||
}
|
||||
if got[0].EarliestInputPosition() != 0 || got[1].EarliestInputPosition() != 2 {
|
||||
t.Fatalf("earliest positions = [%d %d], want [0 2]", got[0].EarliestInputPosition(), got[1].EarliestInputPosition())
|
||||
}
|
||||
|
||||
events := result.AppliedGroups()
|
||||
if len(events) != 2 || len(result.RejectedGroups()) != 0 {
|
||||
t.Fatalf("event counts = applied %d rejected %d, want 2 and 0", len(events), len(result.RejectedGroups()))
|
||||
}
|
||||
first := events[0].Provenance()
|
||||
if !reflect.DeepEqual(first.MemberPositions(), []int{0, 2}) || first.CanonicalPosition() != 2 || !reflect.DeepEqual(first.OriginalInputIndexes(), []int{2, 4}) || first.EarliestInputPosition() != 2 {
|
||||
t.Fatalf("first applied provenance = members %v canonical %d indexes %v earliest %d", first.MemberPositions(), first.CanonicalPosition(), first.OriginalInputIndexes(), first.EarliestInputPosition())
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyPlanWithoutGroupsReturnsOwnedRecordsInProvenanceOrder(t *testing.T) {
|
||||
records := syntheticRecords(t,
|
||||
recordFixture{"alpha", []int{2}, 2},
|
||||
recordFixture{"bravo", []int{0}, 0},
|
||||
recordFixture{"charlie", []int{1}, 1},
|
||||
)
|
||||
result, err := ApplyPlan(Plan{}, records, syntheticPolicy())
|
||||
if err != nil {
|
||||
t.Fatalf("ApplyPlan() error = %v", err)
|
||||
}
|
||||
if names := recordNames(result.Records()); !reflect.DeepEqual(names, []string{"bravo", "charlie", "alpha"}) {
|
||||
t.Fatalf("record names = %v, want [bravo charlie alpha]", names)
|
||||
}
|
||||
if result.AppliedGroups() != nil || result.RejectedGroups() != nil {
|
||||
t.Fatalf("events = applied %#v rejected %#v, want nil", result.AppliedGroups(), result.RejectedGroups())
|
||||
}
|
||||
|
||||
value := result.Records()[0].Value()
|
||||
value.Notes[0] = "changed"
|
||||
if records[1].Value().Notes[0] != "bravo" || result.Records()[0].Value().Notes[0] != "bravo" {
|
||||
t.Fatal("no-group result shares mutable value state")
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyPlanPreservesUngroupedRecordsAndOwnsResults(t *testing.T) {
|
||||
records := syntheticRecords(t,
|
||||
recordFixture{"alpha", []int{4}, 4},
|
||||
recordFixture{"bravo", []int{1}, 1},
|
||||
recordFixture{"charlie", []int{2}, 2},
|
||||
)
|
||||
result, err := ApplyPlan(Plan{groups: []PlanGroup{{memberPositions: []int{0, 2}, canonicalPosition: 0}}}, records, syntheticPolicy())
|
||||
if err != nil {
|
||||
t.Fatalf("ApplyPlan() error = %v", err)
|
||||
}
|
||||
if names := recordNames(result.Records()); !reflect.DeepEqual(names, []string{"bravo", "alpha"}) {
|
||||
t.Fatalf("record names = %v, want ungrouped bravo then consolidated alpha", names)
|
||||
}
|
||||
|
||||
firstRead := result.Records()
|
||||
firstValue := firstRead[0].Value()
|
||||
firstValue.Notes[0] = "changed"
|
||||
firstRead[0].originalInputIndexes[0] = 99
|
||||
if again := result.Records(); again[0].Value().Notes[0] != "bravo" || !reflect.DeepEqual(again[0].OriginalInputIndexes(), []int{1}) {
|
||||
t.Fatalf("result retained accessor mutations: %#v", again[0])
|
||||
}
|
||||
if original := records[1].Value(); original.Notes[0] != "bravo" {
|
||||
t.Fatalf("input record was mutated: %#v", original)
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyPlanGuardRejectionPreservesEveryMemberOnce(t *testing.T) {
|
||||
records := syntheticRecords(t,
|
||||
recordFixture{"alpha", []int{3}, 3},
|
||||
recordFixture{"bravo", []int{1}, 1},
|
||||
recordFixture{"charlie", []int{2}, 2},
|
||||
)
|
||||
consolidations := 0
|
||||
result, err := ApplyPlan(Plan{groups: []PlanGroup{{memberPositions: []int{0, 2}, canonicalPosition: 2}}}, records, ApplicationPolicy[syntheticRecord]{
|
||||
CloneValue: cloneSyntheticRecord,
|
||||
RejectGroup: func(members []syntheticRecord, canonical syntheticRecord) RejectionCategory {
|
||||
members[0].Notes[0] = "guard mutation"
|
||||
canonical.Notes[0] = "canonical mutation"
|
||||
return "typed_constraint"
|
||||
},
|
||||
ConsolidateGroup: func([]syntheticRecord, syntheticRecord) (syntheticRecord, error) {
|
||||
consolidations++
|
||||
return syntheticRecord{}, nil
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("ApplyPlan() error = %v", err)
|
||||
}
|
||||
if consolidations != 0 {
|
||||
t.Fatalf("consolidation calls = %d, want 0", consolidations)
|
||||
}
|
||||
if names := recordNames(result.Records()); !reflect.DeepEqual(names, []string{"bravo", "charlie", "alpha"}) {
|
||||
t.Fatalf("preserved record names = %v, want [bravo charlie alpha]", names)
|
||||
}
|
||||
for index, record := range records {
|
||||
if got := record.Value().Notes[0]; got != record.Value().Name {
|
||||
t.Fatalf("input record %d note = %q after guard, want original", index, got)
|
||||
}
|
||||
}
|
||||
rejected := result.RejectedGroups()
|
||||
if len(rejected) != 1 || rejected[0].Category() != "typed_constraint" {
|
||||
t.Fatalf("rejected events = %#v, want typed_constraint", rejected)
|
||||
}
|
||||
provenance := rejected[0].Provenance()
|
||||
if !reflect.DeepEqual(provenance.MemberPositions(), []int{0, 2}) || !reflect.DeepEqual(provenance.OriginalInputIndexes(), []int{2, 3}) || provenance.EarliestInputPosition() != 2 {
|
||||
t.Fatalf("rejected provenance = members %v indexes %v earliest %d", provenance.MemberPositions(), provenance.OriginalInputIndexes(), provenance.EarliestInputPosition())
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyPlanSuppliesFreshPolicyValuesAndDoesNotMutateInputs(t *testing.T) {
|
||||
records := syntheticRecords(t,
|
||||
recordFixture{"alpha", []int{0}, 0},
|
||||
recordFixture{"bravo", []int{1}, 1},
|
||||
)
|
||||
result, err := ApplyPlan(Plan{groups: []PlanGroup{{memberPositions: []int{0, 1}, canonicalPosition: 1}}}, records, ApplicationPolicy[syntheticRecord]{
|
||||
CloneValue: cloneSyntheticRecord,
|
||||
RejectGroup: func(members []syntheticRecord, canonical syntheticRecord) RejectionCategory {
|
||||
members[0].Name = "guard mutation"
|
||||
canonical.Name = "guard canonical mutation"
|
||||
return ""
|
||||
},
|
||||
ConsolidateGroup: func(members []syntheticRecord, canonical syntheticRecord) (syntheticRecord, error) {
|
||||
if members[0].Name != "alpha" || canonical.Name != "bravo" {
|
||||
t.Fatalf("consolidation observed guard mutations: members %#v canonical %#v", members, canonical)
|
||||
}
|
||||
members[0].Notes[0] = "consolidator mutation"
|
||||
canonical.Notes = []string{"result"}
|
||||
return canonical, nil
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("ApplyPlan() error = %v", err)
|
||||
}
|
||||
if got := result.Records()[0].Value(); got.Name != "bravo" || !reflect.DeepEqual(got.Notes, []string{"result"}) {
|
||||
t.Fatalf("consolidated value = %#v", got)
|
||||
}
|
||||
if records[0].Value().Notes[0] != "alpha" || records[1].Value().Notes[0] != "bravo" {
|
||||
t.Fatalf("input records changed: %#v %#v", records[0].Value(), records[1].Value())
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyPlanRejectsMalformedPlans(t *testing.T) {
|
||||
records := syntheticRecords(t,
|
||||
recordFixture{"alpha", nil, 0},
|
||||
recordFixture{"bravo", nil, 1},
|
||||
recordFixture{"charlie", nil, 2},
|
||||
)
|
||||
tests := []struct {
|
||||
name string
|
||||
plan Plan
|
||||
want string
|
||||
}{
|
||||
{name: "member out of range", plan: Plan{groups: []PlanGroup{{memberPositions: []int{0, 3}, canonicalPosition: 0}}}, want: "outside record range"},
|
||||
{name: "canonical out of range", plan: Plan{groups: []PlanGroup{{memberPositions: []int{0, 1}, canonicalPosition: 3}}}, want: "canonical position"},
|
||||
{name: "canonical not a member", plan: Plan{groups: []PlanGroup{{memberPositions: []int{0, 1}, canonicalPosition: 2}}}, want: "is not a member"},
|
||||
{name: "duplicate member", plan: Plan{groups: []PlanGroup{{memberPositions: []int{0, 0}, canonicalPosition: 0}}}, want: "ascending and unique"},
|
||||
{name: "overlapping groups", plan: Plan{groups: []PlanGroup{{memberPositions: []int{0, 1}, canonicalPosition: 0}, {memberPositions: []int{1, 2}, canonicalPosition: 1}}}, want: "multiple groups"},
|
||||
}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
_, err := ApplyPlan(test.plan, records, syntheticPolicy())
|
||||
if err == nil || !strings.Contains(err.Error(), test.want) {
|
||||
t.Fatalf("ApplyPlan() error = %v, want containing %q", err, test.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyPlanReturnsConsolidationFailures(t *testing.T) {
|
||||
records := syntheticRecords(t,
|
||||
recordFixture{"alpha", nil, 0},
|
||||
recordFixture{"bravo", nil, 1},
|
||||
)
|
||||
want := errors.New("cannot consolidate")
|
||||
result, err := ApplyPlan(Plan{groups: []PlanGroup{{memberPositions: []int{0, 1}, canonicalPosition: 0}}}, records, ApplicationPolicy[syntheticRecord]{
|
||||
CloneValue: cloneSyntheticRecord,
|
||||
ConsolidateGroup: func([]syntheticRecord, syntheticRecord) (syntheticRecord, error) {
|
||||
return syntheticRecord{}, want
|
||||
},
|
||||
})
|
||||
if !errors.Is(err, want) || !strings.Contains(err.Error(), "position 0") {
|
||||
t.Fatalf("ApplyPlan() error = %v, want contextual wrapped failure", err)
|
||||
}
|
||||
if result.Records() != nil || result.AppliedGroups() != nil || result.RejectedGroups() != nil {
|
||||
t.Fatalf("ApplyPlan() partial result = %#v, want zero result", result)
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyPlanPreservesNilAndEmptyRecordCollections(t *testing.T) {
|
||||
policy := syntheticPolicy()
|
||||
nilResult, err := ApplyPlan(Plan{}, []Record[syntheticRecord](nil), policy)
|
||||
if err != nil {
|
||||
t.Fatalf("ApplyPlan(nil) error = %v", err)
|
||||
}
|
||||
emptyResult, err := ApplyPlan(Plan{}, []Record[syntheticRecord]{}, policy)
|
||||
if err != nil {
|
||||
t.Fatalf("ApplyPlan(empty) error = %v", err)
|
||||
}
|
||||
if nilResult.Records() != nil {
|
||||
t.Fatal("nil records became non-nil")
|
||||
}
|
||||
if got := emptyResult.Records(); got == nil || len(got) != 0 {
|
||||
t.Fatalf("empty records = %#v, want non-nil empty", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyPlanValidatesPolicyAndRecordState(t *testing.T) {
|
||||
valid := syntheticPolicy()
|
||||
if _, err := ApplyPlan(Plan{}, []Record[syntheticRecord]{}, ApplicationPolicy[syntheticRecord]{ConsolidateGroup: valid.ConsolidateGroup}); err == nil {
|
||||
t.Fatal("ApplyPlan() accepted nil clone function")
|
||||
}
|
||||
if _, err := ApplyPlan(Plan{}, []Record[syntheticRecord]{}, ApplicationPolicy[syntheticRecord]{CloneValue: cloneSyntheticRecord}); err == nil {
|
||||
t.Fatal("ApplyPlan() accepted nil consolidate function")
|
||||
}
|
||||
if _, err := ApplyPlan(Plan{}, []Record[syntheticRecord]{{}}, valid); err == nil || !strings.Contains(err.Error(), "record 0") {
|
||||
t.Fatalf("ApplyPlan() invalid record error = %v", err)
|
||||
}
|
||||
|
||||
records := syntheticRecords(t,
|
||||
recordFixture{"alpha", nil, 0},
|
||||
recordFixture{"bravo", nil, 1},
|
||||
)
|
||||
valid.RejectGroup = func([]syntheticRecord, syntheticRecord) RejectionCategory { return " " }
|
||||
if _, err := ApplyPlan(Plan{groups: []PlanGroup{{memberPositions: []int{0, 1}, canonicalPosition: 0}}}, records, valid); err == nil || !strings.Contains(err.Error(), "category") {
|
||||
t.Fatalf("ApplyPlan() blank category error = %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
type recordFixture struct {
|
||||
name string
|
||||
indexes []int
|
||||
earliest int
|
||||
}
|
||||
|
||||
func syntheticRecords(t *testing.T, fixtures ...recordFixture) []Record[syntheticRecord] {
|
||||
t.Helper()
|
||||
records := make([]Record[syntheticRecord], len(fixtures))
|
||||
for index, fixture := range fixtures {
|
||||
record, err := NewRecord(syntheticRecord{Name: fixture.name, Notes: []string{fixture.name}}, fixture.indexes, fixture.earliest, cloneSyntheticRecord)
|
||||
if err != nil {
|
||||
t.Fatalf("NewRecord(%d) error = %v", index, err)
|
||||
}
|
||||
records[index] = record
|
||||
}
|
||||
return records
|
||||
}
|
||||
|
||||
func syntheticPolicy() ApplicationPolicy[syntheticRecord] {
|
||||
return ApplicationPolicy[syntheticRecord]{
|
||||
CloneValue: cloneSyntheticRecord,
|
||||
ConsolidateGroup: func(_ []syntheticRecord, canonical syntheticRecord) (syntheticRecord, error) {
|
||||
return canonical, nil
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func recordNames(records []Record[syntheticRecord]) []string {
|
||||
names := make([]string, len(records))
|
||||
for index, record := range records {
|
||||
names[index] = record.Value().Name
|
||||
}
|
||||
return names
|
||||
}
|
||||
111
internal/framework/semanticreconcile/assets.go
Normal file
111
internal/framework/semanticreconcile/assets.go
Normal file
@@ -0,0 +1,111 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"io/fs"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/promptfs"
|
||||
)
|
||||
|
||||
const (
|
||||
PromptID = "generic.semantic_reconciliation"
|
||||
PromptVersion = "v1"
|
||||
promptRoot = "assets/prompts"
|
||||
)
|
||||
|
||||
var promptFiles = []promptfs.ModulePromptFile{
|
||||
{Name: "prompt.yaml", Path: "prompts/prompt.yaml"},
|
||||
{Name: "system.md", Path: "prompts/system.md"},
|
||||
{Name: "protocol.md", Path: "prompts/protocol.md"},
|
||||
{Name: "instructions.md", Path: "prompts/instructions.md"},
|
||||
{Name: "candidates.md", Path: "prompts/candidates.md"},
|
||||
{Name: "transcript-windows.md", Path: "prompts/transcript-windows.md"},
|
||||
}
|
||||
|
||||
// RegisterAssets registers the generic reconciliation prompt and response
|
||||
// schema as one production-owned asset set.
|
||||
func RegisterAssets(registry *llm.AssetRegistry) error {
|
||||
if registry == nil {
|
||||
return fmt.Errorf("semantic reconciliation asset registry must not be nil")
|
||||
}
|
||||
if err := ensureAssetsAbsent(registry); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
assets, err := assetFS()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
prompts, err := promptfs.ModulePromptFS(PromptID, assets, append([]promptfs.ModulePromptFile(nil), promptFiles...))
|
||||
if err != nil {
|
||||
return fmt.Errorf("prepare semantic reconciliation prompt assets: %w", err)
|
||||
}
|
||||
if err := registry.RegisterPromptFS(prompts, promptRoot); err != nil {
|
||||
return fmt.Errorf("register semantic reconciliation prompt assets: %w", err)
|
||||
}
|
||||
if err := registry.RegisterSchemaFS(assets, "schemas"); err != nil {
|
||||
return fmt.Errorf("register semantic reconciliation schema assets: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// PromptHash returns the deterministic identity of the complete generic prompt.
|
||||
func PromptHash() (string, error) {
|
||||
assets, err := assetFS()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
parts := make([]llm.AssetHashPart, 0, len(promptFiles))
|
||||
for _, file := range promptFiles {
|
||||
parts = append(parts, llm.AssetHashPart{FS: assets, Path: file.Path})
|
||||
}
|
||||
return llm.HashAssets(parts)
|
||||
}
|
||||
|
||||
// SchemaHash returns the deterministic identity of the response schema.
|
||||
func SchemaHash() (string, error) {
|
||||
schema, err := LoadResponseSchema()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return schema.SHA256, nil
|
||||
}
|
||||
|
||||
// SharedPromptFiles returns fresh descriptors for the mandatory protocol and
|
||||
// variable-input presentation assets that domain prompts may reuse.
|
||||
func SharedPromptFiles() ([]promptfs.SharedPromptFile, error) {
|
||||
assets, err := assetFS()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return []promptfs.SharedPromptFile{
|
||||
{Name: "protocol.md", FS: assets, Path: "prompts/protocol.md"},
|
||||
{Name: "candidates.md", FS: assets, Path: "prompts/candidates.md"},
|
||||
{Name: "transcript-windows.md", FS: assets, Path: "prompts/transcript-windows.md"},
|
||||
}, nil
|
||||
}
|
||||
|
||||
func ensureAssetsAbsent(registry *llm.AssetRegistry) error {
|
||||
prompts, err := registry.PromptFS()
|
||||
if err != nil {
|
||||
return fmt.Errorf("inspect registered prompt assets: %w", err)
|
||||
}
|
||||
if _, err := fs.Stat(prompts, PromptID+"/prompt.yaml"); err == nil {
|
||||
return fmt.Errorf("semantic reconciliation prompt assets already registered")
|
||||
} else if !errors.Is(err, fs.ErrNotExist) {
|
||||
return fmt.Errorf("inspect semantic reconciliation prompt assets: %w", err)
|
||||
}
|
||||
|
||||
schemas, err := registry.SchemaFS()
|
||||
if err != nil {
|
||||
return fmt.Errorf("inspect registered schema assets: %w", err)
|
||||
}
|
||||
if _, err := fs.Stat(schemas, "semantic_reconciliation_llm.v1.json"); err == nil {
|
||||
return fmt.Errorf("semantic reconciliation schema assets already registered")
|
||||
} else if !errors.Is(err, fs.ErrNotExist) {
|
||||
return fmt.Errorf("inspect semantic reconciliation schema assets: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
118
internal/framework/semanticreconcile/assets_test.go
Normal file
118
internal/framework/semanticreconcile/assets_test.go
Normal file
@@ -0,0 +1,118 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io/fs"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/promptkit"
|
||||
)
|
||||
|
||||
func TestRegisterAssetsPreparesGenericPromptOffline(t *testing.T) {
|
||||
registry := llm.NewAssetRegistry()
|
||||
if err := RegisterAssets(registry); err != nil {
|
||||
t.Fatalf("RegisterAssets() error = %v, want nil", err)
|
||||
}
|
||||
options, err := registry.PromptKitOptions()
|
||||
if err != nil {
|
||||
t.Fatalf("PromptKitOptions() error = %v, want nil", err)
|
||||
}
|
||||
options = append(options, promptkit.WithProfiles(promptkit.OpenAICompatibleProfile(promptkit.OpenAICompatibleProfileConfig{
|
||||
ID: "semantic-reconciliation-test", Endpoint: "http://127.0.0.1:1/v1", Model: "offline-test-model",
|
||||
})))
|
||||
engine, err := promptkit.NewEngine(promptkit.Config{Timeout: time.Second}, options...)
|
||||
if err != nil {
|
||||
t.Fatalf("NewEngine() error = %v, want nil", err)
|
||||
}
|
||||
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{
|
||||
PromptID: PromptID, PromptVersion: PromptVersion, ProfileID: "semantic-reconciliation-test",
|
||||
Inputs: map[string]promptkit.ArtifactRef{
|
||||
"candidates": promptkit.Inline(`{"candidates":[{"candidate_id":1,"label":"Mira"},{"candidate_id":2,"label":"Captain Mira"}]}`),
|
||||
"transcript": promptkit.Inline(`{"windows":[{"units":[{"unit_id":7,"text":"Mira arrived."}]}]}`),
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Prepare() error = %v, want nil", err)
|
||||
}
|
||||
|
||||
if prepared.PromptID != PromptID || prepared.PromptVersion != PromptVersion {
|
||||
t.Fatalf("prepared prompt identity = %q %q, want %q %q", prepared.PromptID, prepared.PromptVersion, PromptID, PromptVersion)
|
||||
}
|
||||
if prepared.SelectedProfileID != "semantic-reconciliation-test" {
|
||||
t.Fatalf("selected profile = %q, want explicit test profile", prepared.SelectedProfileID)
|
||||
}
|
||||
if contract := prepared.OutputContract; contract.SchemaPath != "semantic_reconciliation_llm.v1.json" || contract.RepairAttempts != 0 {
|
||||
t.Fatalf("output contract = %#v, want generic schema without repair", contract)
|
||||
}
|
||||
if len(prepared.Messages) != 5 || prepared.Messages[0].Role != "system" {
|
||||
t.Fatalf("prepared messages = %#v, want five ordered messages beginning with system", prepared.Messages)
|
||||
}
|
||||
if cache := prepared.Messages[2].CacheControl; cache == nil || cache.Type != promptkit.CacheControlEphemeral {
|
||||
t.Fatalf("semantic policy cache control = %#v, want ephemeral", cache)
|
||||
}
|
||||
for _, index := range []int{0, 1, 3, 4} {
|
||||
if prepared.Messages[index].CacheControl != nil {
|
||||
t.Fatalf("message %d cache control = %#v, want nil", index, prepared.Messages[index].CacheControl)
|
||||
}
|
||||
}
|
||||
protocol := prepared.Messages[1].Content
|
||||
for _, requirement := range []string{"positive integer", "Return IDs only", "do not copy candidate names", "source ranges"} {
|
||||
if !strings.Contains(protocol, requirement) {
|
||||
t.Fatalf("protocol message = %q, want requirement %q", protocol, requirement)
|
||||
}
|
||||
}
|
||||
if !strings.Contains(prepared.Messages[3].Content, `"candidate_id":1`) || strings.Contains(prepared.Messages[3].Content, `"windows"`) {
|
||||
t.Fatalf("candidate message = %q, want only integer candidate material", prepared.Messages[3].Content)
|
||||
}
|
||||
if !strings.Contains(prepared.Messages[4].Content, `"windows"`) || strings.Contains(prepared.Messages[4].Content, `"candidate_id"`) {
|
||||
t.Fatalf("transcript message = %q, want only transcript windows", prepared.Messages[4].Content)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssetHashesAreDeterministicAndComplete(t *testing.T) {
|
||||
firstPrompt, err := PromptHash()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
secondPrompt, err := PromptHash()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
schemaHash, err := SchemaHash()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if firstPrompt == "" || firstPrompt != secondPrompt || schemaHash == "" || firstPrompt == schemaHash {
|
||||
t.Fatalf("asset hashes = prompt %q/%q schema %q, want stable distinct hashes", firstPrompt, secondPrompt, schemaHash)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSharedPromptFilesExposeOnlyReusableCoreAssets(t *testing.T) {
|
||||
first, err := SharedPromptFiles()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
second, err := SharedPromptFiles()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
wantNames := []string{"protocol.md", "candidates.md", "transcript-windows.md"}
|
||||
gotNames := make([]string, len(first))
|
||||
for index, file := range first {
|
||||
gotNames[index] = file.Name
|
||||
if content, err := fs.ReadFile(file.FS, file.Path); err != nil || len(content) == 0 {
|
||||
t.Fatalf("shared file %q = %q, %v; want readable content", file.Name, content, err)
|
||||
}
|
||||
}
|
||||
if !reflect.DeepEqual(gotNames, wantNames) {
|
||||
t.Fatalf("shared files = %#v, want narrow allowlist %#v", gotNames, wantNames)
|
||||
}
|
||||
first[0].Name = "changed.md"
|
||||
if second[0].Name != "protocol.md" {
|
||||
t.Fatalf("SharedPromptFiles() reused mutable descriptors: %#v", second)
|
||||
}
|
||||
}
|
||||
3
internal/framework/semanticreconcile/doc.go
Normal file
3
internal/framework/semanticreconcile/doc.go
Normal file
@@ -0,0 +1,3 @@
|
||||
// Package semanticreconcile prepares, executes, and validates bounded
|
||||
// domain-neutral semantic reconciliation requests.
|
||||
package semanticreconcile
|
||||
210
internal/framework/semanticreconcile/engine.go
Normal file
210
internal/framework/semanticreconcile/engine.go
Normal file
@@ -0,0 +1,210 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
)
|
||||
|
||||
// PromptSpec identifies the exact prompt selected by a reconciliation owner.
|
||||
type PromptSpec struct {
|
||||
ID string
|
||||
Version string
|
||||
SHA256 string
|
||||
}
|
||||
|
||||
// Validate rejects incomplete prompt identity and non-canonical digests.
|
||||
func (spec PromptSpec) Validate() error {
|
||||
if strings.TrimSpace(spec.ID) == "" {
|
||||
return fmt.Errorf("semantic reconciliation prompt ID must not be empty")
|
||||
}
|
||||
if strings.TrimSpace(spec.Version) == "" {
|
||||
return fmt.Errorf("semantic reconciliation prompt version must not be empty")
|
||||
}
|
||||
if err := validateSHA256(spec.SHA256); err != nil {
|
||||
return fmt.Errorf("semantic reconciliation prompt digest: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// DefaultPromptSpec returns the identity of the core-owned generic prompt.
|
||||
func DefaultPromptSpec() (PromptSpec, error) {
|
||||
digest, err := PromptHash()
|
||||
if err != nil {
|
||||
return PromptSpec{}, err
|
||||
}
|
||||
return PromptSpec{ID: PromptID, Version: PromptVersion, SHA256: digest}, nil
|
||||
}
|
||||
|
||||
// Request contains one typed owner's source-backed reconciliation input.
|
||||
type Request struct {
|
||||
StageName string
|
||||
Source *source.SourceDocument
|
||||
Candidates []Candidate
|
||||
ProfileID string
|
||||
SessionID string
|
||||
}
|
||||
|
||||
// ResultDisposition classifies a provider-neutral reconciliation outcome.
|
||||
type ResultDisposition uint8
|
||||
|
||||
const (
|
||||
Complete ResultDisposition = iota + 1
|
||||
RetryableInvalidStructuredOutput
|
||||
RetryableDiscardedProposalGroups
|
||||
SkippedInsufficientCandidates
|
||||
SkippedLimitExceeded
|
||||
)
|
||||
|
||||
// Result owns the safe plan and neutral diagnostics from one call.
|
||||
type Result struct {
|
||||
disposition ResultDisposition
|
||||
plan Plan
|
||||
issues []Issue
|
||||
discardedGroupCount int
|
||||
candidateMappings []CandidateMapping
|
||||
}
|
||||
|
||||
// Disposition returns the classified outcome.
|
||||
func (result Result) Disposition() ResultDisposition { return result.disposition }
|
||||
|
||||
// Plan returns an independently owned safe plan.
|
||||
func (result Result) Plan() Plan { return result.planCopy() }
|
||||
|
||||
// Issues returns an owned copy of stable proposal issues.
|
||||
func (result Result) Issues() []Issue { return append([]Issue(nil), result.issues...) }
|
||||
|
||||
// DiscardedGroupCount returns the number of excluded proposal groups.
|
||||
func (result Result) DiscardedGroupCount() int { return result.discardedGroupCount }
|
||||
|
||||
// CandidateMappings returns the request-local handle mapping used for this call.
|
||||
func (result Result) CandidateMappings() []CandidateMapping {
|
||||
return append([]CandidateMapping(nil), result.candidateMappings...)
|
||||
}
|
||||
|
||||
func (result Result) planCopy() Plan {
|
||||
return Plan{groups: result.plan.Groups()}
|
||||
}
|
||||
|
||||
// Engine prepares bounded material, performs one structured completion, and
|
||||
// classifies the deterministic assessment without applying it to typed values.
|
||||
type Engine struct {
|
||||
client contracts.StructuredLLMClient
|
||||
prompt PromptSpec
|
||||
schema llm.ResponseSchema
|
||||
limits Limits
|
||||
}
|
||||
|
||||
// NewEngine constructs a reconciliation engine using the core response schema.
|
||||
func NewEngine(client contracts.StructuredLLMClient, prompt PromptSpec, limits Limits) (*Engine, error) {
|
||||
if client == nil {
|
||||
return nil, fmt.Errorf("construct semantic reconciliation engine: LLM client must not be nil")
|
||||
}
|
||||
if err := prompt.Validate(); err != nil {
|
||||
return nil, fmt.Errorf("construct semantic reconciliation engine: %w", err)
|
||||
}
|
||||
if err := limits.Validate(); err != nil {
|
||||
return nil, fmt.Errorf("construct semantic reconciliation engine: %w", err)
|
||||
}
|
||||
schema, err := LoadResponseSchema()
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("construct semantic reconciliation engine: load response schema: %w", err)
|
||||
}
|
||||
return newEngine(client, prompt, schema, limits), nil
|
||||
}
|
||||
|
||||
func newEngine(client contracts.StructuredLLMClient, prompt PromptSpec, schema llm.ResponseSchema, limits Limits) *Engine {
|
||||
return &Engine{client: client, prompt: prompt, schema: schema, limits: limits}
|
||||
}
|
||||
|
||||
// Reconcile prepares and assesses one request. Retryable semantic outcomes are
|
||||
// returned as results; provider and transport failures remain errors.
|
||||
func (engine *Engine) Reconcile(ctx context.Context, request Request) (Result, error) {
|
||||
if err := engine.validate(); err != nil {
|
||||
return Result{}, err
|
||||
}
|
||||
if ctx == nil {
|
||||
return Result{}, fmt.Errorf("semantic reconciliation %q: context must not be nil", request.StageName)
|
||||
}
|
||||
if strings.TrimSpace(request.StageName) == "" {
|
||||
return Result{}, fmt.Errorf("semantic reconciliation stage name must not be empty")
|
||||
}
|
||||
if request.Source == nil {
|
||||
return Result{}, fmt.Errorf("semantic reconciliation %q: source document must not be nil", request.StageName)
|
||||
}
|
||||
if err := ctx.Err(); err != nil {
|
||||
return Result{}, fmt.Errorf("semantic reconciliation %q: context error before preparation: %w", request.StageName, err)
|
||||
}
|
||||
|
||||
preparation, err := Prepare(request.Source, request.Candidates, engine.limits)
|
||||
if err != nil {
|
||||
return Result{}, fmt.Errorf("semantic reconciliation %q: prepare materials: %w", request.StageName, err)
|
||||
}
|
||||
result := Result{candidateMappings: preparation.CandidateMappings()}
|
||||
switch preparation.Disposition() {
|
||||
case InsufficientCandidates:
|
||||
result.disposition = SkippedInsufficientCandidates
|
||||
return result, nil
|
||||
case LimitExceeded:
|
||||
result.disposition = SkippedLimitExceeded
|
||||
return result, nil
|
||||
case Ready:
|
||||
default:
|
||||
return Result{}, fmt.Errorf("semantic reconciliation %q: unknown preparation disposition %d", request.StageName, preparation.Disposition())
|
||||
}
|
||||
|
||||
if err := ctx.Err(); err != nil {
|
||||
return Result{}, fmt.Errorf("semantic reconciliation %q: context error before completion: %w", request.StageName, err)
|
||||
}
|
||||
var response ProposalResponse
|
||||
_, err = engine.client.CompleteStructured(ctx, contracts.StructuredCompletionRequest{
|
||||
StageName: request.StageName,
|
||||
PromptID: engine.prompt.ID,
|
||||
PromptVersion: engine.prompt.Version,
|
||||
ProfileID: request.ProfileID,
|
||||
SessionID: request.SessionID,
|
||||
Inputs: preparation.Materials(),
|
||||
}, &response)
|
||||
if err != nil {
|
||||
if errors.Is(err, contracts.ErrInvalidStructuredOutput) {
|
||||
result.disposition = RetryableInvalidStructuredOutput
|
||||
return result, nil
|
||||
}
|
||||
return Result{}, fmt.Errorf("semantic reconciliation %q: complete structured output: %w", request.StageName, err)
|
||||
}
|
||||
|
||||
assessment := preparation.Assess(response)
|
||||
result.plan = assessment.Plan()
|
||||
result.issues = assessment.Issues()
|
||||
result.discardedGroupCount = assessment.DiscardedGroupCount()
|
||||
if result.discardedGroupCount > 0 {
|
||||
result.disposition = RetryableDiscardedProposalGroups
|
||||
} else {
|
||||
result.disposition = Complete
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func (engine *Engine) validate() error {
|
||||
if engine == nil {
|
||||
return fmt.Errorf("semantic reconciliation engine must not be nil")
|
||||
}
|
||||
if engine.client == nil {
|
||||
return fmt.Errorf("semantic reconciliation engine: LLM client must not be nil")
|
||||
}
|
||||
if err := engine.prompt.Validate(); err != nil {
|
||||
return fmt.Errorf("semantic reconciliation engine: invalid construction state: %w", err)
|
||||
}
|
||||
if err := engine.limits.Validate(); err != nil {
|
||||
return fmt.Errorf("semantic reconciliation engine: invalid construction state: %w", err)
|
||||
}
|
||||
if err := validateResponseSchemaIdentity(engine.schema); err != nil {
|
||||
return fmt.Errorf("semantic reconciliation engine: invalid construction state: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
304
internal/framework/semanticreconcile/engine_test.go
Normal file
304
internal/framework/semanticreconcile/engine_test.go
Normal file
@@ -0,0 +1,304 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
)
|
||||
|
||||
func TestNewEngineValidatesConstruction(t *testing.T) {
|
||||
prompt := testPromptSpec("a")
|
||||
tests := []struct {
|
||||
name string
|
||||
client contracts.StructuredLLMClient
|
||||
prompt PromptSpec
|
||||
limits Limits
|
||||
want string
|
||||
}{
|
||||
{name: "nil client", prompt: prompt, limits: DefaultLimits(), want: "client"},
|
||||
{name: "empty prompt ID", client: &recordingReconciliationClient{}, prompt: PromptSpec{Version: "v1", SHA256: testDigest("a")}, limits: DefaultLimits(), want: "prompt ID"},
|
||||
{name: "empty prompt version", client: &recordingReconciliationClient{}, prompt: PromptSpec{ID: "prompt", SHA256: testDigest("a")}, limits: DefaultLimits(), want: "prompt version"},
|
||||
{name: "missing digest", client: &recordingReconciliationClient{}, prompt: PromptSpec{ID: "prompt", Version: "v1"}, limits: DefaultLimits(), want: "digest"},
|
||||
{name: "wrong digest algorithm", client: &recordingReconciliationClient{}, prompt: PromptSpec{ID: "prompt", Version: "v1", SHA256: "md5:" + strings.Repeat("a", 32)}, limits: DefaultLimits(), want: "sha256:"},
|
||||
{name: "non canonical digest", client: &recordingReconciliationClient{}, prompt: PromptSpec{ID: "prompt", Version: "v1", SHA256: "sha256:" + strings.Repeat("A", 64)}, limits: DefaultLimits(), want: "lowercase"},
|
||||
{name: "invalid limits", client: &recordingReconciliationClient{}, prompt: prompt, limits: Limits{}, want: "limits"},
|
||||
}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
if _, err := NewEngine(test.client, test.prompt, test.limits); err == nil || !strings.Contains(err.Error(), test.want) {
|
||||
t.Fatalf("NewEngine() error = %v, want %q", err, test.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
defaultPrompt, err := DefaultPromptSpec()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if defaultPrompt.ID != PromptID || defaultPrompt.Version != PromptVersion || defaultPrompt.SHA256 == "" {
|
||||
t.Fatalf("DefaultPromptSpec() = %#v, want complete generic prompt identity", defaultPrompt)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEnginePropagatesRequestAndAssessesResponse(t *testing.T) {
|
||||
client := &recordingReconciliationClient{responses: []ProposalResponse{{DuplicateGroups: []DuplicateGroup{{
|
||||
CandidateIDs: []int{1, 2}, CanonicalCandidateID: 2,
|
||||
}}}}}
|
||||
engine := newTestEngine(t, client, DefaultLimits())
|
||||
request := readyEngineRequest()
|
||||
request.ProfileID = " profile-as-resolved "
|
||||
request.SessionID = " session-as-supplied "
|
||||
|
||||
result, err := engine.Reconcile(context.Background(), request)
|
||||
if err != nil {
|
||||
t.Fatalf("Reconcile() error = %v, want nil", err)
|
||||
}
|
||||
if result.Disposition() != Complete || result.DiscardedGroupCount() != 0 || len(result.Issues()) != 0 {
|
||||
t.Fatalf("result = disposition %v discarded %d issues %#v", result.Disposition(), result.DiscardedGroupCount(), result.Issues())
|
||||
}
|
||||
groups := result.Plan().Groups()
|
||||
if len(groups) != 1 || !reflect.DeepEqual(groups[0].MemberPositions(), []int{0, 1}) || groups[0].CanonicalPosition() != 1 {
|
||||
t.Fatalf("safe plan = %#v", groups)
|
||||
}
|
||||
if len(client.requests) != 1 {
|
||||
t.Fatalf("completion calls = %d, want exactly one", len(client.requests))
|
||||
}
|
||||
got := client.requests[0]
|
||||
if got.StageName != request.StageName || got.PromptID != engine.prompt.ID || got.PromptVersion != engine.prompt.Version || got.ProfileID != request.ProfileID || got.SessionID != request.SessionID {
|
||||
t.Fatalf("structured request = %#v, want exact routing values", got)
|
||||
}
|
||||
if len(got.Inputs) != 2 || got.Inputs["candidates"].Name != "candidates" || got.Inputs["transcript"].Name != "transcript" || len(got.Vars) != 0 {
|
||||
t.Fatalf("structured request inputs = %#v vars = %#v, want only candidate and transcript materials", got.Inputs, got.Vars)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEngineClassifiesSemanticAndTransportOutcomes(t *testing.T) {
|
||||
transportErr := errors.New("provider unavailable")
|
||||
tests := []struct {
|
||||
name string
|
||||
response ProposalResponse
|
||||
completion error
|
||||
want ResultDisposition
|
||||
wantDiscard int
|
||||
wantIssues bool
|
||||
wantError error
|
||||
}{
|
||||
{name: "empty groups complete", response: ProposalResponse{DuplicateGroups: []DuplicateGroup{}}, want: Complete},
|
||||
{name: "discarded proposal retryable", response: ProposalResponse{DuplicateGroups: []DuplicateGroup{{CandidateIDs: []int{1, 99}, CanonicalCandidateID: 1}}}, want: RetryableDiscardedProposalGroups, wantDiscard: 1, wantIssues: true},
|
||||
{name: "invalid structured output retryable", completion: fmt.Errorf("decode response: %w", contracts.ErrInvalidStructuredOutput), want: RetryableInvalidStructuredOutput},
|
||||
{name: "transport failure", completion: transportErr, wantError: transportErr},
|
||||
}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
client := &recordingReconciliationClient{responses: []ProposalResponse{test.response}, errors: []error{test.completion}}
|
||||
result, err := newTestEngine(t, client, DefaultLimits()).Reconcile(context.Background(), readyEngineRequest())
|
||||
if test.wantError != nil {
|
||||
if !errors.Is(err, test.wantError) || !strings.Contains(err.Error(), readyEngineRequest().StageName) {
|
||||
t.Fatalf("Reconcile() error = %v, want contextual %v", err, test.wantError)
|
||||
}
|
||||
return
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("Reconcile() error = %v, want nil", err)
|
||||
}
|
||||
if result.Disposition() != test.want || result.DiscardedGroupCount() != test.wantDiscard || (len(result.Issues()) > 0) != test.wantIssues {
|
||||
t.Fatalf("result = disposition %v discarded %d issues %#v", result.Disposition(), result.DiscardedGroupCount(), result.Issues())
|
||||
}
|
||||
if len(client.requests) != 1 {
|
||||
t.Fatalf("completion calls = %d, want one", len(client.requests))
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestEngineSkipsDeterministicOutcomesWithoutCompletion(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
request Request
|
||||
limits Limits
|
||||
want ResultDisposition
|
||||
mappingLen int
|
||||
}{
|
||||
{name: "insufficient candidates", request: engineRequestWithCandidateCount(1), limits: DefaultLimits(), want: SkippedInsufficientCandidates, mappingLen: 1},
|
||||
{name: "candidate limit", request: engineRequestWithCandidateCount(2), limits: Limits{ContextRadius: 0, MaximumCandidates: 1, MaximumMaterialBytes: 10000}, want: SkippedLimitExceeded, mappingLen: 2},
|
||||
{name: "material limit", request: engineRequestWithCandidateCount(2), limits: Limits{ContextRadius: 0, MaximumCandidates: 2, MaximumMaterialBytes: 1}, want: SkippedLimitExceeded, mappingLen: 2},
|
||||
}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
client := &recordingReconciliationClient{}
|
||||
result, err := newTestEngine(t, client, test.limits).Reconcile(context.Background(), test.request)
|
||||
if err != nil {
|
||||
t.Fatalf("Reconcile() error = %v, want nil", err)
|
||||
}
|
||||
if result.Disposition() != test.want || len(result.CandidateMappings()) != test.mappingLen {
|
||||
t.Fatalf("result disposition = %v mappings = %#v", result.Disposition(), result.CandidateMappings())
|
||||
}
|
||||
if len(client.requests) != 0 {
|
||||
t.Fatalf("completion calls = %d, want zero", len(client.requests))
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestEngineRejectsInvalidInvocationAndHonorsCancellation(t *testing.T) {
|
||||
request := readyEngineRequest()
|
||||
client := &recordingReconciliationClient{}
|
||||
engine := newTestEngine(t, client, DefaultLimits())
|
||||
|
||||
var nilEngine *Engine
|
||||
if _, err := nilEngine.Reconcile(context.Background(), request); err == nil || !strings.Contains(err.Error(), "engine must not be nil") {
|
||||
t.Fatalf("nil engine error = %v", err)
|
||||
}
|
||||
if _, err := (&Engine{}).Reconcile(context.Background(), request); err == nil || !strings.Contains(err.Error(), "client") {
|
||||
t.Fatalf("zero engine error = %v", err)
|
||||
}
|
||||
invalid := newEngine(&recordingReconciliationClient{}, testPromptSpec("a"), llm.ResponseSchema{}, DefaultLimits())
|
||||
if _, err := invalid.Reconcile(context.Background(), request); err == nil || !strings.Contains(err.Error(), "invalid construction state") {
|
||||
t.Fatalf("invalid construction error = %v", err)
|
||||
}
|
||||
if _, err := engine.Reconcile(nil, request); err == nil || !strings.Contains(err.Error(), "context") {
|
||||
t.Fatalf("nil context error = %v", err)
|
||||
}
|
||||
withoutStage := request
|
||||
withoutStage.StageName = " "
|
||||
if _, err := engine.Reconcile(context.Background(), withoutStage); err == nil || !strings.Contains(err.Error(), "stage name") {
|
||||
t.Fatalf("empty stage error = %v", err)
|
||||
}
|
||||
withoutSource := request
|
||||
withoutSource.Source = nil
|
||||
if _, err := engine.Reconcile(context.Background(), withoutSource); err == nil || !strings.Contains(err.Error(), "source document") {
|
||||
t.Fatalf("nil source error = %v", err)
|
||||
}
|
||||
|
||||
canceled, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
if _, err := engine.Reconcile(canceled, request); !errors.Is(err, context.Canceled) || !strings.Contains(err.Error(), "before preparation") {
|
||||
t.Fatalf("preparation cancellation error = %v", err)
|
||||
}
|
||||
if _, err := engine.Reconcile(&cancelBeforeCompletionContext{}, request); !errors.Is(err, context.Canceled) || !strings.Contains(err.Error(), "before completion") {
|
||||
t.Fatalf("completion cancellation error = %v", err)
|
||||
}
|
||||
if len(client.requests) != 0 {
|
||||
t.Fatalf("completion calls = %d, want zero for invalid and canceled invocations", len(client.requests))
|
||||
}
|
||||
}
|
||||
|
||||
func TestEngineCallsAreIndependentAndResultsAreOwned(t *testing.T) {
|
||||
client := &recordingReconciliationClient{responses: []ProposalResponse{
|
||||
{DuplicateGroups: []DuplicateGroup{
|
||||
{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 1},
|
||||
{CandidateIDs: []int{2, 99}, CanonicalCandidateID: 2},
|
||||
}},
|
||||
{DuplicateGroups: []DuplicateGroup{}},
|
||||
}}
|
||||
engine := newTestEngine(t, client, DefaultLimits())
|
||||
first, err := engine.Reconcile(context.Background(), readyEngineRequest())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
first.Plan().Groups()[0].memberPositions[0] = 99
|
||||
firstIssues := first.Issues()
|
||||
firstIssues[0].Category = "changed"
|
||||
firstMappings := first.CandidateMappings()
|
||||
firstMappings[0].CandidatePosition = 99
|
||||
if first.Plan().Groups()[0].MemberPositions()[0] != 0 || first.Issues()[0].Category == "changed" || first.CandidateMappings()[0].CandidatePosition != 0 {
|
||||
t.Fatal("result accessors exposed retained state")
|
||||
}
|
||||
|
||||
second, err := engine.Reconcile(context.Background(), readyEngineRequest())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if second.Disposition() != Complete || len(second.Plan().Groups()) != 0 || len(second.Issues()) != 0 || second.DiscardedGroupCount() != 0 || len(second.CandidateMappings()) != 2 {
|
||||
t.Fatalf("second result retained prior call state: disposition %v plan %#v issues %#v discarded %d mappings %#v", second.Disposition(), second.Plan().Groups(), second.Issues(), second.DiscardedGroupCount(), second.CandidateMappings())
|
||||
}
|
||||
}
|
||||
|
||||
type recordingReconciliationClient struct {
|
||||
requests []contracts.StructuredCompletionRequest
|
||||
responses []ProposalResponse
|
||||
errors []error
|
||||
}
|
||||
|
||||
func (client *recordingReconciliationClient) CompleteStructured(_ context.Context, request contracts.StructuredCompletionRequest, output any) (contracts.StructuredCompletionResponse, error) {
|
||||
request.Inputs = request.Inputs.Clone()
|
||||
client.requests = append(client.requests, request)
|
||||
index := len(client.requests) - 1
|
||||
if index < len(client.errors) && client.errors[index] != nil {
|
||||
return contracts.StructuredCompletionResponse{}, client.errors[index]
|
||||
}
|
||||
response := ProposalResponse{DuplicateGroups: []DuplicateGroup{}}
|
||||
if index < len(client.responses) {
|
||||
response = cloneProposalResponse(client.responses[index])
|
||||
}
|
||||
target, ok := output.(*ProposalResponse)
|
||||
if !ok {
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("output type = %T", output)
|
||||
}
|
||||
*target = response
|
||||
return contracts.StructuredCompletionResponse{}, nil
|
||||
}
|
||||
|
||||
func cloneProposalResponse(response ProposalResponse) ProposalResponse {
|
||||
cloned := ProposalResponse{DuplicateGroups: make([]DuplicateGroup, len(response.DuplicateGroups))}
|
||||
for index, group := range response.DuplicateGroups {
|
||||
cloned.DuplicateGroups[index] = DuplicateGroup{
|
||||
CandidateIDs: append([]int(nil), group.CandidateIDs...), CanonicalCandidateID: group.CanonicalCandidateID,
|
||||
}
|
||||
}
|
||||
return cloned
|
||||
}
|
||||
|
||||
func newTestEngine(t *testing.T, client contracts.StructuredLLMClient, limits Limits) *Engine {
|
||||
t.Helper()
|
||||
engine, err := NewEngine(client, testPromptSpec("a"), limits)
|
||||
if err != nil {
|
||||
t.Fatalf("NewEngine() error = %v", err)
|
||||
}
|
||||
return engine
|
||||
}
|
||||
|
||||
func testPromptSpec(digestCharacter string) PromptSpec {
|
||||
return PromptSpec{ID: "test.semantic_reconciliation", Version: "v1", SHA256: testDigest(digestCharacter)}
|
||||
}
|
||||
|
||||
func testDigest(character string) string { return "sha256:" + strings.Repeat(character, 64) }
|
||||
|
||||
func readyEngineRequest() Request { return engineRequestWithCandidateCount(2) }
|
||||
|
||||
func engineRequestWithCandidateCount(count int) Request {
|
||||
document := &source.SourceDocument{ID: "source", Units: []source.SourceUnit{
|
||||
{ID: 1, Kind: "speech", Text: "Mira arrived."},
|
||||
{ID: 2, Kind: "speech", Text: "The captain spoke."},
|
||||
}}
|
||||
candidates := make([]Candidate, count)
|
||||
for index := range candidates {
|
||||
unitID := index%len(document.Units) + 1
|
||||
candidates[index] = Candidate{
|
||||
Label: fmt.Sprintf("candidate-%d", index+1),
|
||||
SourceRefs: []source.SourceRef{{SourceID: document.ID, StartUnitID: unitID, EndUnitID: unitID}},
|
||||
}
|
||||
}
|
||||
return Request{StageName: "test/normalize", Source: document, Candidates: candidates, ProfileID: "profile", SessionID: "session"}
|
||||
}
|
||||
|
||||
type cancelBeforeCompletionContext struct{ calls int }
|
||||
|
||||
func (ctx *cancelBeforeCompletionContext) Deadline() (time.Time, bool) { return time.Time{}, false }
|
||||
func (ctx *cancelBeforeCompletionContext) Done() <-chan struct{} { return nil }
|
||||
func (ctx *cancelBeforeCompletionContext) Value(any) any { return nil }
|
||||
func (ctx *cancelBeforeCompletionContext) Err() error {
|
||||
ctx.calls++
|
||||
if ctx.calls > 1 {
|
||||
return context.Canceled
|
||||
}
|
||||
return nil
|
||||
}
|
||||
101
internal/framework/semanticreconcile/identity.go
Normal file
101
internal/framework/semanticreconcile/identity.go
Normal file
@@ -0,0 +1,101 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
)
|
||||
|
||||
// Policy identifies the framework-owned reconciliation and assessment rules.
|
||||
const Policy = "semantic_reconciliation.v1"
|
||||
|
||||
var _ contracts.ManifestMetadataProvider = (*Engine)(nil)
|
||||
var _ pipeline.CheckpointFingerprintProvider = (*Engine)(nil)
|
||||
|
||||
// ManifestMetadata returns fresh, content-free identity for the complete core
|
||||
// reconciliation mechanism.
|
||||
func (engine *Engine) ManifestMetadata() map[string]any {
|
||||
if engine == nil || engine.validate() != nil {
|
||||
return nil
|
||||
}
|
||||
return map[string]any{
|
||||
"prompt_id": engine.prompt.ID,
|
||||
"prompt_version": engine.prompt.Version,
|
||||
"prompt_sha256": engine.prompt.SHA256,
|
||||
"response_schema_key": string(engine.schema.Key),
|
||||
"response_schema_id": engine.schema.ID,
|
||||
"response_schema_name": engine.schema.Name,
|
||||
"response_schema_version": engine.schema.Version,
|
||||
"response_schema_sha256": engine.schema.SHA256,
|
||||
"semantic_reconciliation_policy": Policy,
|
||||
"semantic_reconciliation_limits": map[string]any{
|
||||
"context_radius": engine.limits.ContextRadius,
|
||||
"maximum_candidates": engine.limits.MaximumCandidates,
|
||||
"maximum_material_bytes": engine.limits.MaximumMaterialBytes,
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// CheckpointFingerprints returns fresh canonical identities for prompt,
|
||||
// schema, reconciliation policy, and the complete limit policy.
|
||||
func (engine *Engine) CheckpointFingerprints() []pipeline.CheckpointFingerprint {
|
||||
if engine == nil || engine.validate() != nil {
|
||||
return nil
|
||||
}
|
||||
return []pipeline.CheckpointFingerprint{
|
||||
{Name: "prompt", Value: identityDigest(engine.prompt.ID, engine.prompt.Version, engine.prompt.SHA256)},
|
||||
{Name: "response_schema", Value: identityDigest(string(engine.schema.Key), engine.schema.ID, engine.schema.Version, engine.schema.Name, engine.schema.SHA256)},
|
||||
{Name: "semantic_reconciliation_policy", Value: Policy},
|
||||
{Name: "semantic_reconciliation_limits", Value: limitPolicyDigest(engine.limits)},
|
||||
}
|
||||
}
|
||||
|
||||
func limitPolicyDigest(limits Limits) string {
|
||||
return identityDigest(
|
||||
strconv.Itoa(limits.ContextRadius),
|
||||
strconv.Itoa(limits.MaximumCandidates),
|
||||
strconv.Itoa(limits.MaximumMaterialBytes),
|
||||
)
|
||||
}
|
||||
|
||||
func identityDigest(parts ...string) string {
|
||||
hash := sha256.New()
|
||||
for _, part := range parts {
|
||||
_, _ = hash.Write([]byte(strconv.Itoa(len(part))))
|
||||
_, _ = hash.Write([]byte{':'})
|
||||
_, _ = hash.Write([]byte(part))
|
||||
}
|
||||
return "sha256:" + hex.EncodeToString(hash.Sum(nil))
|
||||
}
|
||||
|
||||
func validateResponseSchemaIdentity(schema llm.ResponseSchema) error {
|
||||
if strings.TrimSpace(string(schema.Key)) == "" || strings.TrimSpace(schema.ID) == "" || strings.TrimSpace(schema.Version) == "" || strings.TrimSpace(schema.Name) == "" {
|
||||
return fmt.Errorf("response schema identity must be complete")
|
||||
}
|
||||
if err := validateSHA256(schema.SHA256); err != nil {
|
||||
return fmt.Errorf("response schema digest: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func validateSHA256(value string) error {
|
||||
const prefix = "sha256:"
|
||||
if !strings.HasPrefix(value, prefix) {
|
||||
return fmt.Errorf("must use sha256: prefix")
|
||||
}
|
||||
hexValue := strings.TrimPrefix(value, prefix)
|
||||
if len(hexValue) != sha256.Size*2 || hexValue != strings.ToLower(hexValue) {
|
||||
return fmt.Errorf("must contain 64 lowercase hexadecimal characters")
|
||||
}
|
||||
decoded, err := hex.DecodeString(hexValue)
|
||||
if err != nil || len(decoded) != sha256.Size {
|
||||
return fmt.Errorf("must contain 64 lowercase hexadecimal characters")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
93
internal/framework/semanticreconcile/identity_test.go
Normal file
93
internal/framework/semanticreconcile/identity_test.go
Normal file
@@ -0,0 +1,93 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
)
|
||||
|
||||
func TestEngineMetadataAndFingerprintsCoverCoreIdentity(t *testing.T) {
|
||||
base := newTestEngine(t, &recordingReconciliationClient{}, DefaultLimits())
|
||||
metadata := base.ManifestMetadata()
|
||||
for _, key := range []string{
|
||||
"prompt_id", "prompt_version", "prompt_sha256",
|
||||
"response_schema_key", "response_schema_id", "response_schema_name", "response_schema_version", "response_schema_sha256",
|
||||
"semantic_reconciliation_policy", "semantic_reconciliation_limits",
|
||||
} {
|
||||
if metadata[key] == nil || metadata[key] == "" {
|
||||
t.Fatalf("metadata[%q] = %#v, want populated core identity", key, metadata[key])
|
||||
}
|
||||
}
|
||||
limits, ok := metadata["semantic_reconciliation_limits"].(map[string]any)
|
||||
if !ok || len(limits) != 3 || limits["context_radius"] == nil || limits["maximum_candidates"] == nil || limits["maximum_material_bytes"] == nil {
|
||||
t.Fatalf("limit metadata = %#v, want complete limits", metadata["semantic_reconciliation_limits"])
|
||||
}
|
||||
fingerprints := base.CheckpointFingerprints()
|
||||
wantNames := []string{"prompt", "response_schema", "semantic_reconciliation_policy", "semantic_reconciliation_limits"}
|
||||
if len(fingerprints) != len(wantNames) {
|
||||
t.Fatalf("fingerprints = %#v, want required categories", fingerprints)
|
||||
}
|
||||
for index, want := range wantNames {
|
||||
if fingerprints[index].Name != want || fingerprints[index].Value == "" {
|
||||
t.Fatalf("fingerprint %d = %#v, want %q with value", index, fingerprints[index], want)
|
||||
}
|
||||
}
|
||||
|
||||
metadata["prompt_id"] = "changed"
|
||||
limits["context_radius"] = -1
|
||||
fingerprints[0].Name = "changed"
|
||||
if got := base.ManifestMetadata(); got["prompt_id"] == "changed" || got["semantic_reconciliation_limits"].(map[string]any)["context_radius"] == -1 {
|
||||
t.Fatalf("ManifestMetadata() exposed retained state: %#v", got)
|
||||
}
|
||||
if got := base.CheckpointFingerprints(); got[0].Name == "changed" {
|
||||
t.Fatalf("CheckpointFingerprints() exposed retained state: %#v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestCoreFingerprintsChangeWithBehavioralIdentity(t *testing.T) {
|
||||
base := newTestEngine(t, &recordingReconciliationClient{}, DefaultLimits())
|
||||
baseFingerprints := base.CheckpointFingerprints()
|
||||
|
||||
promptChanged, err := NewEngine(&recordingReconciliationClient{}, testPromptSpec("b"), DefaultLimits())
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
assertOnlyFingerprintChanged(t, baseFingerprints, promptChanged.CheckpointFingerprints(), "prompt")
|
||||
|
||||
schema, err := LoadResponseSchema()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
schema.SHA256 = testDigest("b")
|
||||
schemaChanged := newEngine(&recordingReconciliationClient{}, base.prompt, schema, DefaultLimits())
|
||||
assertOnlyFingerprintChanged(t, baseFingerprints, schemaChanged.CheckpointFingerprints(), "response_schema")
|
||||
|
||||
limits := DefaultLimits()
|
||||
limits.ContextRadius++
|
||||
limitsChanged, err := NewEngine(&recordingReconciliationClient{}, base.prompt, limits)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
assertOnlyFingerprintChanged(t, baseFingerprints, limitsChanged.CheckpointFingerprints(), "semantic_reconciliation_limits")
|
||||
}
|
||||
|
||||
func TestInvalidEngineIdentityHasNoMetadataOrFingerprints(t *testing.T) {
|
||||
invalid := newEngine(&recordingReconciliationClient{}, testPromptSpec("a"), llm.ResponseSchema{}, DefaultLimits())
|
||||
if invalid.ManifestMetadata() != nil || invalid.CheckpointFingerprints() != nil {
|
||||
t.Fatalf("invalid engine exposed identity: metadata %#v fingerprints %#v", invalid.ManifestMetadata(), invalid.CheckpointFingerprints())
|
||||
}
|
||||
}
|
||||
|
||||
func assertOnlyFingerprintChanged(t *testing.T, before, after []pipeline.CheckpointFingerprint, changedName string) {
|
||||
t.Helper()
|
||||
if len(before) != len(after) {
|
||||
t.Fatalf("fingerprint counts differ: %#v %#v", before, after)
|
||||
}
|
||||
for index := range before {
|
||||
changed := before[index] != after[index]
|
||||
if changed != (before[index].Name == changedName) {
|
||||
t.Fatalf("fingerprint %q change = %t, want only %q changed\nbefore: %#v\nafter: %#v", before[index].Name, changed, changedName, before, after)
|
||||
}
|
||||
}
|
||||
}
|
||||
403
internal/framework/semanticreconcile/preparation.go
Normal file
403
internal/framework/semanticreconcile/preparation.go
Normal file
@@ -0,0 +1,403 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"sort"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
)
|
||||
|
||||
const (
|
||||
candidateInputName = "candidates"
|
||||
transcriptInputName = "transcript"
|
||||
jsonMediaType = "application/json"
|
||||
)
|
||||
|
||||
var defaultLimits = Limits{
|
||||
ContextRadius: 2,
|
||||
MaximumCandidates: 128,
|
||||
MaximumMaterialBytes: 262144,
|
||||
}
|
||||
|
||||
// Candidate is contextual source-backed input supplied by a typed consumer.
|
||||
// Prepare does not retain or mutate Label or SourceRefs.
|
||||
type Candidate struct {
|
||||
Label string
|
||||
SourceRefs []source.SourceRef
|
||||
}
|
||||
|
||||
// Limits bounds source context and serialized model input.
|
||||
type Limits struct {
|
||||
ContextRadius int
|
||||
MaximumCandidates int
|
||||
MaximumMaterialBytes int
|
||||
}
|
||||
|
||||
// DefaultLimits returns the core-owned production limits.
|
||||
func DefaultLimits() Limits {
|
||||
return defaultLimits
|
||||
}
|
||||
|
||||
// Validate rejects limits that cannot safely bound preparation.
|
||||
func (limits Limits) Validate() error {
|
||||
if limits.ContextRadius < 0 {
|
||||
return fmt.Errorf("semantic reconciliation limits: context radius must not be negative")
|
||||
}
|
||||
if limits.MaximumCandidates <= 0 {
|
||||
return fmt.Errorf("semantic reconciliation limits: maximum candidates must be positive")
|
||||
}
|
||||
if limits.MaximumMaterialBytes <= 0 {
|
||||
return fmt.Errorf("semantic reconciliation limits: maximum bytes must be positive")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Disposition describes whether prepared materials may be sent to a model.
|
||||
type Disposition uint8
|
||||
|
||||
const (
|
||||
// Ready indicates that the result contains complete bounded materials.
|
||||
Ready Disposition = iota + 1
|
||||
// InsufficientCandidates indicates that fewer than two candidates were
|
||||
// eligible after source-reference validation.
|
||||
InsufficientCandidates
|
||||
// LimitExceeded indicates that a candidate or serialized-material bound was
|
||||
// exceeded and no request should be split or sent.
|
||||
LimitExceeded
|
||||
)
|
||||
|
||||
// CandidateMapping relates one model-visible request-local ID to the
|
||||
// corresponding zero-based position in the caller's candidate slice.
|
||||
type CandidateMapping struct {
|
||||
CandidateID int
|
||||
CandidatePosition int
|
||||
}
|
||||
|
||||
// Preparation owns the visible candidate mapping and prompt materials.
|
||||
type Preparation struct {
|
||||
disposition Disposition
|
||||
mappings []CandidateMapping
|
||||
materials contracts.LLMInputSet
|
||||
}
|
||||
|
||||
// Disposition returns the preparation outcome.
|
||||
func (preparation Preparation) Disposition() Disposition {
|
||||
return preparation.disposition
|
||||
}
|
||||
|
||||
// CandidateMappings returns an owned copy in model-visible candidate order.
|
||||
func (preparation Preparation) CandidateMappings() []CandidateMapping {
|
||||
return append([]CandidateMapping(nil), preparation.mappings...)
|
||||
}
|
||||
|
||||
// Materials returns independently owned candidate and transcript materials.
|
||||
// It is empty unless Disposition returns Ready.
|
||||
func (preparation Preparation) Materials() contracts.LLMInputSet {
|
||||
return preparation.materials.Clone()
|
||||
}
|
||||
|
||||
type sourceRange struct {
|
||||
StartUnitID int `json:"start_unit_id"`
|
||||
EndUnitID int `json:"end_unit_id"`
|
||||
}
|
||||
|
||||
type visibleCandidate struct {
|
||||
CandidateID int `json:"candidate_id"`
|
||||
Label string `json:"label"`
|
||||
SourceRefs []sourceRange `json:"source_refs"`
|
||||
}
|
||||
|
||||
type candidateInput struct {
|
||||
Candidates []visibleCandidate `json:"candidates"`
|
||||
}
|
||||
|
||||
type transcriptInput struct {
|
||||
Windows []transcriptWindow `json:"windows"`
|
||||
}
|
||||
|
||||
type transcriptWindow struct {
|
||||
Units []transcriptUnit `json:"units"`
|
||||
}
|
||||
|
||||
type transcriptUnit struct {
|
||||
ID int `json:"id"`
|
||||
Kind string `json:"kind"`
|
||||
Text string `json:"text"`
|
||||
Metadata map[string]any `json:"metadata,omitempty"`
|
||||
Cited bool `json:"cited"`
|
||||
}
|
||||
|
||||
type sourceInterval struct {
|
||||
start int
|
||||
end int
|
||||
}
|
||||
|
||||
type preparedCandidate struct {
|
||||
position int
|
||||
references []sourceRange
|
||||
intervals []sourceInterval
|
||||
}
|
||||
|
||||
// Prepare validates candidates and constructs bounded, source-ordered model
|
||||
// inputs. Deterministic skip conditions are represented by the returned
|
||||
// disposition rather than an error.
|
||||
func Prepare(document *source.SourceDocument, candidates []Candidate, limits Limits) (Preparation, error) {
|
||||
if err := limits.Validate(); err != nil {
|
||||
return Preparation{}, err
|
||||
}
|
||||
|
||||
documentIndex := source.NewDocumentIndex(document)
|
||||
prepared := make([]preparedCandidate, 0, len(candidates))
|
||||
for candidatePosition, candidate := range candidates {
|
||||
references, intervals, valid := prepareReferences(documentIndex, candidate.SourceRefs)
|
||||
if !valid {
|
||||
continue
|
||||
}
|
||||
prepared = append(prepared, preparedCandidate{
|
||||
position: candidatePosition,
|
||||
references: references,
|
||||
intervals: intervals,
|
||||
})
|
||||
}
|
||||
|
||||
result := Preparation{
|
||||
disposition: InsufficientCandidates,
|
||||
mappings: make([]CandidateMapping, len(prepared)),
|
||||
}
|
||||
views := make([]visibleCandidate, len(prepared))
|
||||
for index, candidate := range prepared {
|
||||
candidateID := index + 1
|
||||
result.mappings[index] = CandidateMapping{
|
||||
CandidateID: candidateID,
|
||||
CandidatePosition: candidate.position,
|
||||
}
|
||||
views[index] = visibleCandidate{
|
||||
CandidateID: candidateID,
|
||||
Label: candidates[candidate.position].Label,
|
||||
SourceRefs: cloneSourceRanges(candidate.references),
|
||||
}
|
||||
}
|
||||
if len(prepared) < 2 {
|
||||
return result, nil
|
||||
}
|
||||
if len(prepared) > limits.MaximumCandidates {
|
||||
result.disposition = LimitExceeded
|
||||
return result, nil
|
||||
}
|
||||
|
||||
candidateContent, withinLimit, err := marshalCandidateInput(views, limits.MaximumMaterialBytes)
|
||||
if err != nil {
|
||||
return Preparation{}, fmt.Errorf("prepare semantic reconciliation: encode candidate material: %w", err)
|
||||
}
|
||||
if !withinLimit {
|
||||
result.disposition = LimitExceeded
|
||||
return result, nil
|
||||
}
|
||||
|
||||
contextIntervals := make([]sourceInterval, 0)
|
||||
citedIntervals := make([]sourceInterval, 0)
|
||||
for _, candidate := range prepared {
|
||||
for _, interval := range candidate.intervals {
|
||||
citedIntervals = append(citedIntervals, interval)
|
||||
contextIntervals = append(contextIntervals, sourceInterval{
|
||||
start: max(0, interval.start-limits.ContextRadius),
|
||||
end: min(len(document.Units)-1, interval.end+limits.ContextRadius),
|
||||
})
|
||||
}
|
||||
}
|
||||
transcriptContent, withinLimit, err := marshalTranscriptInput(
|
||||
document.Units,
|
||||
coalesceIntervals(contextIntervals),
|
||||
coalesceIntervals(citedIntervals),
|
||||
limits.MaximumMaterialBytes-len(candidateContent),
|
||||
)
|
||||
if err != nil {
|
||||
return Preparation{}, fmt.Errorf("prepare semantic reconciliation: build transcript material: invalid source metadata")
|
||||
}
|
||||
if !withinLimit {
|
||||
result.disposition = LimitExceeded
|
||||
return result, nil
|
||||
}
|
||||
|
||||
result.disposition = Ready
|
||||
result.materials = contracts.LLMInputSet{
|
||||
candidateInputName: newInputMaterial(candidateInputName, candidateContent),
|
||||
transcriptInputName: newInputMaterial(transcriptInputName, transcriptContent),
|
||||
}
|
||||
return result, nil
|
||||
}
|
||||
|
||||
func prepareReferences(index source.DocumentIndex, references []source.SourceRef) ([]sourceRange, []sourceInterval, bool) {
|
||||
if len(references) == 0 {
|
||||
return nil, nil, false
|
||||
}
|
||||
type referencedInterval struct {
|
||||
reference sourceRange
|
||||
interval sourceInterval
|
||||
}
|
||||
prepared := make([]referencedInterval, 0, len(references))
|
||||
for _, reference := range references {
|
||||
if err := index.ValidateRef(reference); err != nil {
|
||||
return nil, nil, false
|
||||
}
|
||||
start, _ := index.Position(reference.StartUnitID)
|
||||
end, _ := index.Position(reference.EndUnitID)
|
||||
prepared = append(prepared, referencedInterval{
|
||||
reference: sourceRange{StartUnitID: reference.StartUnitID, EndUnitID: reference.EndUnitID},
|
||||
interval: sourceInterval{start: start, end: end},
|
||||
})
|
||||
}
|
||||
sort.Slice(prepared, func(left, right int) bool {
|
||||
if prepared[left].interval.start != prepared[right].interval.start {
|
||||
return prepared[left].interval.start < prepared[right].interval.start
|
||||
}
|
||||
return prepared[left].interval.end < prepared[right].interval.end
|
||||
})
|
||||
|
||||
canonicalReferences := make([]sourceRange, 0, len(prepared))
|
||||
intervals := make([]sourceInterval, 0, len(prepared))
|
||||
for _, item := range prepared {
|
||||
if len(canonicalReferences) > 0 && canonicalReferences[len(canonicalReferences)-1] == item.reference {
|
||||
continue
|
||||
}
|
||||
canonicalReferences = append(canonicalReferences, item.reference)
|
||||
intervals = append(intervals, item.interval)
|
||||
}
|
||||
return canonicalReferences, intervals, true
|
||||
}
|
||||
|
||||
func cloneSourceRanges(ranges []sourceRange) []sourceRange {
|
||||
if len(ranges) == 0 {
|
||||
return []sourceRange{}
|
||||
}
|
||||
return append([]sourceRange(nil), ranges...)
|
||||
}
|
||||
|
||||
func coalesceIntervals(intervals []sourceInterval) []sourceInterval {
|
||||
if len(intervals) == 0 {
|
||||
return nil
|
||||
}
|
||||
ordered := append([]sourceInterval(nil), intervals...)
|
||||
sort.Slice(ordered, func(left, right int) bool {
|
||||
if ordered[left].start != ordered[right].start {
|
||||
return ordered[left].start < ordered[right].start
|
||||
}
|
||||
return ordered[left].end < ordered[right].end
|
||||
})
|
||||
|
||||
coalesced := make([]sourceInterval, 0, len(ordered))
|
||||
for _, interval := range ordered {
|
||||
if len(coalesced) == 0 || interval.start > coalesced[len(coalesced)-1].end+1 {
|
||||
coalesced = append(coalesced, interval)
|
||||
continue
|
||||
}
|
||||
if interval.end > coalesced[len(coalesced)-1].end {
|
||||
coalesced[len(coalesced)-1].end = interval.end
|
||||
}
|
||||
}
|
||||
return coalesced
|
||||
}
|
||||
|
||||
func marshalCandidateInput(candidates []visibleCandidate, maximumBytes int) ([]byte, bool, error) {
|
||||
content := make([]byte, 0, min(maximumBytes, 4096))
|
||||
var withinLimit bool
|
||||
content, withinLimit = appendWithinLimit(content, maximumBytes, []byte(`{"candidates":[`))
|
||||
if !withinLimit {
|
||||
return nil, false, nil
|
||||
}
|
||||
for index, candidate := range candidates {
|
||||
encoded, err := json.Marshal(candidate)
|
||||
if err != nil {
|
||||
return nil, false, err
|
||||
}
|
||||
separator := []byte(nil)
|
||||
if index > 0 {
|
||||
separator = []byte(",")
|
||||
}
|
||||
content, withinLimit = appendWithinLimit(content, maximumBytes, separator, encoded)
|
||||
if !withinLimit {
|
||||
return nil, false, nil
|
||||
}
|
||||
}
|
||||
content, withinLimit = appendWithinLimit(content, maximumBytes, []byte("]}"))
|
||||
return content, withinLimit, nil
|
||||
}
|
||||
|
||||
// marshalTranscriptInput retains at most maximumBytes while visiting source
|
||||
// units in order. It deliberately serializes one unit at a time so an oversized
|
||||
// request does not require a document-sized transcript copy before rejection.
|
||||
func marshalTranscriptInput(units []source.SourceUnit, contextIntervals, citedIntervals []sourceInterval, maximumBytes int) ([]byte, bool, error) {
|
||||
content := make([]byte, 0, min(maximumBytes, 4096))
|
||||
content, withinLimit := appendWithinLimit(content, maximumBytes, []byte(`{"windows":[`))
|
||||
if !withinLimit {
|
||||
return nil, false, nil
|
||||
}
|
||||
|
||||
citedIndex := 0
|
||||
for windowIndex, interval := range contextIntervals {
|
||||
separator := []byte(nil)
|
||||
if windowIndex > 0 {
|
||||
separator = []byte(",")
|
||||
}
|
||||
content, withinLimit = appendWithinLimit(content, maximumBytes, separator, []byte(`{"units":[`))
|
||||
if !withinLimit {
|
||||
return nil, false, nil
|
||||
}
|
||||
for position := interval.start; position <= interval.end; position++ {
|
||||
for citedIndex < len(citedIntervals) && citedIntervals[citedIndex].end < position {
|
||||
citedIndex++
|
||||
}
|
||||
cited := citedIndex < len(citedIntervals) && citedIntervals[citedIndex].start <= position
|
||||
unit := units[position]
|
||||
metadata, err := source.CloneMetadata(unit.Metadata)
|
||||
if err != nil {
|
||||
return nil, false, err
|
||||
}
|
||||
encoded, err := json.Marshal(transcriptUnit{
|
||||
ID: unit.ID, Kind: unit.Kind, Text: unit.Text, Metadata: metadata, Cited: cited,
|
||||
})
|
||||
if err != nil {
|
||||
return nil, false, err
|
||||
}
|
||||
separator = nil
|
||||
if position > interval.start {
|
||||
separator = []byte(",")
|
||||
}
|
||||
content, withinLimit = appendWithinLimit(content, maximumBytes, separator, encoded)
|
||||
if !withinLimit {
|
||||
return nil, false, nil
|
||||
}
|
||||
}
|
||||
content, withinLimit = appendWithinLimit(content, maximumBytes, []byte("]}"))
|
||||
if !withinLimit {
|
||||
return nil, false, nil
|
||||
}
|
||||
}
|
||||
content, withinLimit = appendWithinLimit(content, maximumBytes, []byte("]}"))
|
||||
return content, withinLimit, nil
|
||||
}
|
||||
|
||||
func appendWithinLimit(content []byte, maximumBytes int, parts ...[]byte) ([]byte, bool) {
|
||||
for _, part := range parts {
|
||||
if len(content) > maximumBytes || len(part) > maximumBytes-len(content) {
|
||||
return content, false
|
||||
}
|
||||
content = append(content, part...)
|
||||
}
|
||||
return content, true
|
||||
}
|
||||
|
||||
func newInputMaterial(name string, content []byte) contracts.LLMInputMaterial {
|
||||
digest := sha256.Sum256(content)
|
||||
return contracts.NewLLMInputMaterial(
|
||||
name,
|
||||
jsonMediaType,
|
||||
content,
|
||||
"sha256:"+hex.EncodeToString(digest[:]),
|
||||
"",
|
||||
)
|
||||
}
|
||||
386
internal/framework/semanticreconcile/preparation_test.go
Normal file
386
internal/framework/semanticreconcile/preparation_test.go
Normal file
@@ -0,0 +1,386 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"math"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
)
|
||||
|
||||
func TestPrepareBuildsContiguousCandidatesAndOwnedSourceContext(t *testing.T) {
|
||||
document := &source.SourceDocument{ID: "private-source-id", Units: []source.SourceUnit{
|
||||
{ID: 40, Kind: "narration", Text: "zero"},
|
||||
{ID: 10, Kind: "speech", Text: "one", Metadata: map[string]any{"speaker": map[string]any{"name": "Mira"}}},
|
||||
{ID: 70, Kind: "speech", Text: "two"},
|
||||
{ID: 20, Kind: "narration", Text: "three"},
|
||||
{ID: 90, Kind: "speech", Text: "four"},
|
||||
}}
|
||||
references := []source.SourceRef{
|
||||
{SourceID: document.ID, StartUnitID: 90, EndUnitID: 90},
|
||||
{SourceID: document.ID, StartUnitID: 10, EndUnitID: 20},
|
||||
{SourceID: document.ID, StartUnitID: 10, EndUnitID: 20},
|
||||
}
|
||||
candidates := []Candidate{
|
||||
{Label: "The Tavern", SourceRefs: append([]source.SourceRef(nil), references...)},
|
||||
{Label: "The Tavern", SourceRefs: append([]source.SourceRef(nil), references...)},
|
||||
{Label: "Broken", SourceRefs: []source.SourceRef{{SourceID: document.ID, StartUnitID: 20, EndUnitID: 10}}},
|
||||
}
|
||||
before := cloneCandidates(candidates)
|
||||
|
||||
preparation, err := Prepare(document, candidates, Limits{
|
||||
ContextRadius: 1,
|
||||
MaximumCandidates: len(candidates),
|
||||
MaximumMaterialBytes: 10000,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if preparation.Disposition() != Ready {
|
||||
t.Fatalf("Disposition() = %v, want Ready", preparation.Disposition())
|
||||
}
|
||||
if !reflect.DeepEqual(candidates, before) {
|
||||
t.Fatalf("Prepare() mutated candidates: %#v", candidates)
|
||||
}
|
||||
if got, want := preparation.CandidateMappings(), []CandidateMapping{
|
||||
{CandidateID: 1, CandidatePosition: 0},
|
||||
{CandidateID: 2, CandidatePosition: 1},
|
||||
}; !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("CandidateMappings() = %#v, want %#v", got, want)
|
||||
}
|
||||
|
||||
materials := preparation.Materials()
|
||||
if len(materials) != 2 {
|
||||
t.Fatalf("materials = %#v, want candidates and transcript", materials)
|
||||
}
|
||||
if _, ok := materials[candidateInputName]; !ok {
|
||||
t.Fatal("candidate material is missing")
|
||||
}
|
||||
if _, ok := materials[transcriptInputName]; !ok {
|
||||
t.Fatal("transcript material is missing")
|
||||
}
|
||||
for name, material := range materials {
|
||||
if material.Name != name || material.MediaType != "application/json" || material.OriginURI != "" || material.SizeBytes != int64(len(material.Content)) {
|
||||
t.Fatalf("material %q metadata = %#v", name, material)
|
||||
}
|
||||
digest := sha256.Sum256(material.Content)
|
||||
if want := "sha256:" + hex.EncodeToString(digest[:]); material.Digest != want {
|
||||
t.Fatalf("material %q digest = %q, want %q", name, material.Digest, want)
|
||||
}
|
||||
}
|
||||
|
||||
candidateContent := append([]byte(nil), materials[candidateInputName].Content...)
|
||||
var candidatePayload candidateInput
|
||||
if err := json.Unmarshal(candidateContent, &candidatePayload); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
wantCandidates := []visibleCandidate{
|
||||
{CandidateID: 1, Label: "The Tavern", SourceRefs: []sourceRange{{StartUnitID: 10, EndUnitID: 20}, {StartUnitID: 90, EndUnitID: 90}}},
|
||||
{CandidateID: 2, Label: "The Tavern", SourceRefs: []sourceRange{{StartUnitID: 10, EndUnitID: 20}, {StartUnitID: 90, EndUnitID: 90}}},
|
||||
}
|
||||
if !reflect.DeepEqual(candidatePayload.Candidates, wantCandidates) {
|
||||
t.Fatalf("candidate payload = %#v, want %#v", candidatePayload.Candidates, wantCandidates)
|
||||
}
|
||||
var candidateObjects struct {
|
||||
Candidates []map[string]json.RawMessage `json:"candidates"`
|
||||
}
|
||||
if err := json.Unmarshal(candidateContent, &candidateObjects); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, candidate := range candidateObjects.Candidates {
|
||||
if len(candidate) != 3 || candidate["candidate_id"] == nil || candidate["label"] == nil || candidate["source_refs"] == nil {
|
||||
t.Fatalf("model-facing candidate fields = %#v", candidate)
|
||||
}
|
||||
}
|
||||
combined := string(materials[candidateInputName].Content) + string(materials[transcriptInputName].Content)
|
||||
for _, forbidden := range []string{document.ID, "application_entity_id", "private-entity-id"} {
|
||||
if strings.Contains(combined, forbidden) {
|
||||
t.Fatalf("model material leaked %q: %s", forbidden, combined)
|
||||
}
|
||||
}
|
||||
|
||||
var transcript transcriptInput
|
||||
if err := json.Unmarshal(materials[transcriptInputName].Content, &transcript); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(transcript.Windows) != 1 || len(transcript.Windows[0].Units) != len(document.Units) {
|
||||
t.Fatalf("windows = %#v, want one coalesced source window", transcript.Windows)
|
||||
}
|
||||
for index, wantID := range []int{40, 10, 70, 20, 90} {
|
||||
if transcript.Windows[0].Units[index].ID != wantID {
|
||||
t.Fatalf("unit %d id = %d, want %d", index, transcript.Windows[0].Units[index].ID, wantID)
|
||||
}
|
||||
}
|
||||
if transcript.Windows[0].Units[0].Cited {
|
||||
t.Fatal("radius-only unit marked cited")
|
||||
}
|
||||
for index := 1; index < len(transcript.Windows[0].Units); index++ {
|
||||
if !transcript.Windows[0].Units[index].Cited {
|
||||
t.Fatalf("evidence unit %d was not marked cited", index)
|
||||
}
|
||||
}
|
||||
|
||||
candidates[0].SourceRefs[0].StartUnitID = 40
|
||||
document.Units[1].Metadata["speaker"].(map[string]any)["name"] = "changed"
|
||||
if got := preparation.Materials()[candidateInputName].Content; !reflect.DeepEqual(got, candidateContent) {
|
||||
t.Fatalf("candidate material changed through caller input: %s", got)
|
||||
}
|
||||
var retained transcriptInput
|
||||
if err := json.Unmarshal(preparation.Materials()[transcriptInputName].Content, &retained); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := retained.Windows[0].Units[1].Metadata["speaker"].(map[string]any)["name"]; got != "Mira" {
|
||||
t.Fatalf("retained metadata = %v, want Mira", got)
|
||||
}
|
||||
|
||||
returnedMappings := preparation.CandidateMappings()
|
||||
returnedMappings[0].CandidatePosition = 99
|
||||
returnedMaterials := preparation.Materials()
|
||||
candidateMaterial := returnedMaterials[candidateInputName]
|
||||
candidateMaterial.Content[0] = '['
|
||||
returnedMaterials[candidateInputName] = candidateMaterial
|
||||
delete(returnedMaterials, transcriptInputName)
|
||||
if preparation.CandidateMappings()[0].CandidatePosition != 0 || !json.Valid(preparation.Materials()[candidateInputName].Content) || len(preparation.Materials()) != 2 {
|
||||
t.Fatal("preparation accessors exposed retained data")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrepareRedactsInvalidSourceMetadata(t *testing.T) {
|
||||
const sensitiveKey = "sensitive-metadata-key"
|
||||
document := &source.SourceDocument{ID: "private-source", Units: []source.SourceUnit{
|
||||
{ID: 1, Text: "private transcript", Metadata: map[string]any{sensitiveKey: math.NaN()}},
|
||||
{ID: 2, Text: "other private transcript"},
|
||||
}}
|
||||
candidates := []Candidate{
|
||||
{Label: "Private One", SourceRefs: []source.SourceRef{{SourceID: document.ID, StartUnitID: 1, EndUnitID: 1}}},
|
||||
{Label: "Private Two", SourceRefs: []source.SourceRef{{SourceID: document.ID, StartUnitID: 2, EndUnitID: 2}}},
|
||||
}
|
||||
_, err := Prepare(document, candidates, DefaultLimits())
|
||||
if err == nil || !strings.Contains(err.Error(), "invalid source metadata") {
|
||||
t.Fatalf("Prepare() error = %v, want redacted metadata failure", err)
|
||||
}
|
||||
for _, forbidden := range []string{sensitiveKey, document.ID, document.Units[0].Text, candidates[0].Label, "non-finite", "float64"} {
|
||||
if strings.Contains(err.Error(), forbidden) {
|
||||
t.Fatalf("Prepare() error leaked %q: %v", forbidden, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrepareFiltersUnsafeCandidatesAndCoalescesAdjacentWindows(t *testing.T) {
|
||||
document := &source.SourceDocument{ID: "session", Units: []source.SourceUnit{
|
||||
{ID: 9}, {ID: 3}, {ID: 8}, {ID: 1}, {ID: 7},
|
||||
}}
|
||||
candidates := []Candidate{
|
||||
{Label: "One", SourceRefs: []source.SourceRef{{SourceID: document.ID, StartUnitID: 3, EndUnitID: 3}}},
|
||||
{Label: "Two", SourceRefs: []source.SourceRef{{SourceID: document.ID, StartUnitID: 8, EndUnitID: 8}}},
|
||||
{Label: "Missing", SourceRefs: []source.SourceRef{{SourceID: document.ID, StartUnitID: 99, EndUnitID: 99}}},
|
||||
{Label: "Foreign", SourceRefs: []source.SourceRef{{SourceID: "other", StartUnitID: 1, EndUnitID: 1}}},
|
||||
{Label: "No references"},
|
||||
{Label: "Partly invalid", SourceRefs: []source.SourceRef{
|
||||
{SourceID: document.ID, StartUnitID: 1, EndUnitID: 1},
|
||||
{SourceID: document.ID, StartUnitID: 100, EndUnitID: 100},
|
||||
}},
|
||||
}
|
||||
limits := Limits{ContextRadius: 0, MaximumCandidates: len(candidates), MaximumMaterialBytes: 10000}
|
||||
|
||||
preparation, err := Prepare(document, candidates, limits)
|
||||
if err != nil || preparation.Disposition() != Ready {
|
||||
t.Fatalf("Prepare() disposition = %v, error = %v", preparation.Disposition(), err)
|
||||
}
|
||||
if got, want := preparation.CandidateMappings(), []CandidateMapping{
|
||||
{CandidateID: 1, CandidatePosition: 0},
|
||||
{CandidateID: 2, CandidatePosition: 1},
|
||||
}; !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("CandidateMappings() = %#v, want %#v", got, want)
|
||||
}
|
||||
var transcript transcriptInput
|
||||
if err := json.Unmarshal(preparation.Materials()[transcriptInputName].Content, &transcript); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(transcript.Windows) != 1 || len(transcript.Windows[0].Units) != 2 || transcript.Windows[0].Units[0].ID != 3 || transcript.Windows[0].Units[1].ID != 8 {
|
||||
t.Fatalf("windows = %#v, want adjacent source-order units coalesced", transcript.Windows)
|
||||
}
|
||||
oneCandidate, err := Prepare(document, candidates[:1], limits)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got, want := oneCandidate.CandidateMappings(), []CandidateMapping{{CandidateID: 1, CandidatePosition: 0}}; oneCandidate.Disposition() != InsufficientCandidates || !reflect.DeepEqual(got, want) || len(oneCandidate.Materials()) != 0 {
|
||||
t.Fatalf("Prepare(one candidate) = disposition %v, mappings %#v, materials %#v", oneCandidate.Disposition(), got, oneCandidate.Materials())
|
||||
}
|
||||
|
||||
nilPreparation, err := Prepare(nil, candidates, limits)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if nilPreparation.Disposition() != InsufficientCandidates || len(nilPreparation.CandidateMappings()) != 0 || len(nilPreparation.Materials()) != 0 {
|
||||
t.Fatalf("Prepare(nil) = disposition %v, mappings %#v, materials %#v", nilPreparation.Disposition(), nilPreparation.CandidateMappings(), nilPreparation.Materials())
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrepareValidatesLimitsBeforeBuildingMaterials(t *testing.T) {
|
||||
if err := DefaultLimits().Validate(); err != nil {
|
||||
t.Fatalf("DefaultLimits().Validate() error = %v", err)
|
||||
}
|
||||
cycle := map[string]any{}
|
||||
cycle["self"] = cycle
|
||||
document := &source.SourceDocument{ID: "session", Units: []source.SourceUnit{
|
||||
{ID: 1, Metadata: cycle}, {ID: 2},
|
||||
}}
|
||||
candidates := candidatesForEveryUnit(document)
|
||||
tests := []struct {
|
||||
name string
|
||||
limits Limits
|
||||
want string
|
||||
}{
|
||||
{name: "negative radius", limits: Limits{ContextRadius: -1, MaximumCandidates: 2, MaximumMaterialBytes: 100}, want: "radius"},
|
||||
{name: "zero candidates", limits: Limits{ContextRadius: 0, MaximumCandidates: 0, MaximumMaterialBytes: 100}, want: "candidates"},
|
||||
{name: "negative candidates", limits: Limits{ContextRadius: 0, MaximumCandidates: -1, MaximumMaterialBytes: 100}, want: "candidates"},
|
||||
{name: "zero bytes", limits: Limits{ContextRadius: 0, MaximumCandidates: 2, MaximumMaterialBytes: 0}, want: "bytes"},
|
||||
{name: "negative bytes", limits: Limits{ContextRadius: 0, MaximumCandidates: 2, MaximumMaterialBytes: -1}, want: "bytes"},
|
||||
}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
if _, err := Prepare(document, candidates, test.limits); err == nil || !strings.Contains(err.Error(), test.want) {
|
||||
t.Fatalf("Prepare() error = %v, want %q validation", err, test.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrepareEnforcesCandidateLimitBeforeRenderingContext(t *testing.T) {
|
||||
cycle := map[string]any{}
|
||||
cycle["self"] = cycle
|
||||
document := &source.SourceDocument{ID: "session", Units: []source.SourceUnit{
|
||||
{ID: 1, Text: "one", Metadata: cycle},
|
||||
{ID: 2, Text: "two"},
|
||||
{ID: 3, Text: "three"},
|
||||
}}
|
||||
candidates := candidatesForEveryUnit(document)
|
||||
limits := Limits{ContextRadius: 0, MaximumCandidates: 2, MaximumMaterialBytes: 10000}
|
||||
|
||||
exceeded, err := Prepare(document, candidates, limits)
|
||||
if err != nil {
|
||||
t.Fatalf("Prepare(over limit) error = %v; context should not be rendered", err)
|
||||
}
|
||||
if exceeded.Disposition() != LimitExceeded || len(exceeded.CandidateMappings()) != 3 || len(exceeded.Materials()) != 0 {
|
||||
t.Fatalf("Prepare(over limit) = disposition %v, mappings %#v, materials %#v", exceeded.Disposition(), exceeded.CandidateMappings(), exceeded.Materials())
|
||||
}
|
||||
|
||||
document.Units[0].Metadata = nil
|
||||
exact, err := Prepare(document, candidates[:2], limits)
|
||||
if err != nil || exact.Disposition() != Ready {
|
||||
t.Fatalf("Prepare(at limit) = disposition %v, error %v", exact.Disposition(), err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrepareStopsRenderingContextWhenMaterialLimitIsExceeded(t *testing.T) {
|
||||
cycle := map[string]any{}
|
||||
cycle["self"] = cycle
|
||||
document := &source.SourceDocument{ID: "session", Units: []source.SourceUnit{
|
||||
{ID: 1, Text: strings.Repeat("oversized", 100)},
|
||||
{ID: 2, Text: "must not be inspected"},
|
||||
}}
|
||||
candidates := candidatesForEveryUnit(document)
|
||||
base, err := Prepare(document, candidates, Limits{
|
||||
ContextRadius: 0,
|
||||
MaximumCandidates: len(candidates),
|
||||
MaximumMaterialBytes: 10000,
|
||||
})
|
||||
if err != nil || base.Disposition() != Ready {
|
||||
t.Fatalf("Prepare(base) = disposition %v, error %v", base.Disposition(), err)
|
||||
}
|
||||
candidateBytes := len(base.Materials()[candidateInputName].Content)
|
||||
document.Units[1].Metadata = cycle
|
||||
|
||||
limited, err := Prepare(document, candidates, Limits{
|
||||
ContextRadius: 0,
|
||||
MaximumCandidates: len(candidates),
|
||||
MaximumMaterialBytes: candidateBytes + 64,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Prepare(limited) error = %v; rendering should stop at the material bound", err)
|
||||
}
|
||||
if limited.Disposition() != LimitExceeded || len(limited.Materials()) != 0 {
|
||||
t.Fatalf("Prepare(limited) = disposition %v, materials %#v, want bounded skip", limited.Disposition(), limited.Materials())
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrepareAcceptsExactCombinedByteLimitAndSkipsOneOver(t *testing.T) {
|
||||
document := &source.SourceDocument{ID: "session", Units: []source.SourceUnit{
|
||||
{ID: 1, Text: "one"},
|
||||
{ID: 2, Text: "two"},
|
||||
}}
|
||||
candidates := candidatesForEveryUnit(document)
|
||||
baseLimits := Limits{ContextRadius: 0, MaximumCandidates: len(candidates), MaximumMaterialBytes: 10000}
|
||||
base, err := Prepare(document, candidates, baseLimits)
|
||||
if err != nil || base.Disposition() != Ready {
|
||||
t.Fatalf("Prepare(base) = disposition %v, error %v", base.Disposition(), err)
|
||||
}
|
||||
materials := base.Materials()
|
||||
totalBytes := len(materials[candidateInputName].Content) + len(materials[transcriptInputName].Content)
|
||||
|
||||
exactLimits := baseLimits
|
||||
exactLimits.MaximumMaterialBytes = totalBytes
|
||||
exact, err := Prepare(document, candidates, exactLimits)
|
||||
if err != nil || exact.Disposition() != Ready {
|
||||
t.Fatalf("Prepare(exact bytes) = disposition %v, error %v", exact.Disposition(), err)
|
||||
}
|
||||
oneOverLimits := exactLimits
|
||||
oneOverLimits.MaximumMaterialBytes--
|
||||
oneOver, err := Prepare(document, candidates, oneOverLimits)
|
||||
if err != nil || oneOver.Disposition() != LimitExceeded || len(oneOver.Materials()) != 0 {
|
||||
t.Fatalf("Prepare(one over) = disposition %v, materials %#v, error %v", oneOver.Disposition(), oneOver.Materials(), err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPrepareSerializationIsDeterministic(t *testing.T) {
|
||||
document := &source.SourceDocument{ID: "session", Units: []source.SourceUnit{
|
||||
{ID: 5, Text: "five", Metadata: map[string]any{"z": 1, "a": []any{"first", "second"}}},
|
||||
{ID: 2, Text: "two", Metadata: map[string]any{"nested": map[string]any{"b": true, "a": false}}},
|
||||
}}
|
||||
candidates := candidatesForEveryUnit(document)
|
||||
limits := Limits{ContextRadius: 0, MaximumCandidates: len(candidates), MaximumMaterialBytes: 10000}
|
||||
|
||||
first, err := Prepare(document, candidates, limits)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
second, err := Prepare(document, cloneCandidates(candidates), limits)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
for _, name := range []string{candidateInputName, transcriptInputName} {
|
||||
firstMaterial := first.Materials()[name]
|
||||
secondMaterial := second.Materials()[name]
|
||||
if !reflect.DeepEqual(firstMaterial, secondMaterial) {
|
||||
t.Fatalf("material %q is not deterministic:\n%#v\n%#v", name, firstMaterial, secondMaterial)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func candidatesForEveryUnit(document *source.SourceDocument) []Candidate {
|
||||
candidates := make([]Candidate, len(document.Units))
|
||||
for index, unit := range document.Units {
|
||||
candidates[index] = Candidate{
|
||||
Label: "candidate",
|
||||
SourceRefs: []source.SourceRef{{
|
||||
SourceID: document.ID,
|
||||
StartUnitID: unit.ID,
|
||||
EndUnitID: unit.ID,
|
||||
}},
|
||||
}
|
||||
}
|
||||
return candidates
|
||||
}
|
||||
|
||||
func cloneCandidates(candidates []Candidate) []Candidate {
|
||||
cloned := append([]Candidate(nil), candidates...)
|
||||
for index := range cloned {
|
||||
cloned[index].SourceRefs = append([]source.SourceRef(nil), candidates[index].SourceRefs...)
|
||||
}
|
||||
return cloned
|
||||
}
|
||||
236
internal/framework/semanticreconcile/proposal.go
Normal file
236
internal/framework/semanticreconcile/proposal.go
Normal file
@@ -0,0 +1,236 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
)
|
||||
|
||||
// ProposalResponse is the complete private structured response contract.
|
||||
type ProposalResponse struct {
|
||||
DuplicateGroups []DuplicateGroup `json:"duplicate_groups"`
|
||||
}
|
||||
|
||||
// DuplicateGroup proposes supplied request-local candidate IDs that may denote
|
||||
// one entity and identifies one supplied member as canonical.
|
||||
type DuplicateGroup struct {
|
||||
CandidateIDs []int `json:"candidate_ids"`
|
||||
CanonicalCandidateID int `json:"canonical_candidate_id"`
|
||||
}
|
||||
|
||||
// IssueCategory identifies one stable proposal safety failure.
|
||||
type IssueCategory string
|
||||
|
||||
const (
|
||||
IssueMemberNonPositive IssueCategory = "member_non_positive"
|
||||
IssueMemberUnknown IssueCategory = "member_unknown"
|
||||
IssueRepeatedMember IssueCategory = "repeated_member"
|
||||
IssueFewerThanTwoMembers IssueCategory = "fewer_than_two_members"
|
||||
IssueCanonicalNonPositive IssueCategory = "canonical_non_positive"
|
||||
IssueCanonicalUnknown IssueCategory = "canonical_unknown"
|
||||
IssueCanonicalNotMember IssueCategory = "canonical_not_member"
|
||||
IssueOverlappingMember IssueCategory = "overlapping_member"
|
||||
)
|
||||
|
||||
// Issue identifies an unsafe proposal category at its original response group
|
||||
// index without prescribing caller warning text.
|
||||
type Issue struct {
|
||||
GroupIndex int
|
||||
Category IssueCategory
|
||||
}
|
||||
|
||||
// IssueDetails renders stable, domain-neutral proposal diagnostics for an
|
||||
// adapter's retry message.
|
||||
func IssueDetails(issues []Issue) []string {
|
||||
details := make([]string, len(issues))
|
||||
for index, issue := range issues {
|
||||
details[index] = fmt.Sprintf("group %d: %s", issue.GroupIndex, issue.Category)
|
||||
}
|
||||
return details
|
||||
}
|
||||
|
||||
// PlanGroup identifies one validated group using original candidate positions.
|
||||
type PlanGroup struct {
|
||||
memberPositions []int
|
||||
canonicalPosition int
|
||||
}
|
||||
|
||||
// MemberPositions returns an owned, ascending list of original candidate
|
||||
// positions.
|
||||
func (group PlanGroup) MemberPositions() []int {
|
||||
return append([]int(nil), group.memberPositions...)
|
||||
}
|
||||
|
||||
// CanonicalPosition returns the original position of the selected canonical
|
||||
// candidate.
|
||||
func (group PlanGroup) CanonicalPosition() int {
|
||||
return group.canonicalPosition
|
||||
}
|
||||
|
||||
// Plan contains deterministic, non-overlapping reconciliation groups.
|
||||
type Plan struct {
|
||||
groups []PlanGroup
|
||||
}
|
||||
|
||||
// Groups returns a deeply owned copy ordered by each group's earliest member.
|
||||
func (plan Plan) Groups() []PlanGroup {
|
||||
groups := make([]PlanGroup, len(plan.groups))
|
||||
for index, group := range plan.groups {
|
||||
groups[index] = clonePlanGroup(group)
|
||||
}
|
||||
return groups
|
||||
}
|
||||
|
||||
// Assessment contains the safe plan and stable diagnostics for discarded
|
||||
// response groups.
|
||||
type Assessment struct {
|
||||
plan Plan
|
||||
discardedGroupCount int
|
||||
issues []Issue
|
||||
}
|
||||
|
||||
// Plan returns an independently owned reconciliation plan.
|
||||
func (assessment Assessment) Plan() Plan {
|
||||
groups := assessment.plan.Groups()
|
||||
return Plan{groups: groups}
|
||||
}
|
||||
|
||||
// Issues returns an owned copy ordered by original response group index.
|
||||
func (assessment Assessment) Issues() []Issue {
|
||||
return append([]Issue(nil), assessment.issues...)
|
||||
}
|
||||
|
||||
// DiscardedGroupCount returns the number of response groups excluded from the
|
||||
// safe plan.
|
||||
func (assessment Assessment) DiscardedGroupCount() int {
|
||||
return assessment.discardedGroupCount
|
||||
}
|
||||
|
||||
// RetryRequired reports whether any response group was discarded.
|
||||
func (assessment Assessment) RetryRequired() bool {
|
||||
return assessment.discardedGroupCount > 0
|
||||
}
|
||||
|
||||
type assessedGroup struct {
|
||||
memberPositions []int
|
||||
canonicalPosition int
|
||||
issues []IssueCategory
|
||||
locallyValid bool
|
||||
conflicting bool
|
||||
}
|
||||
|
||||
// Assess resolves request-local IDs through the retained preparation mapping
|
||||
// and returns only deterministic, non-overlapping groups.
|
||||
func (preparation Preparation) Assess(response ProposalResponse) Assessment {
|
||||
positionsByID := make(map[int]int, len(preparation.mappings))
|
||||
for _, mapping := range preparation.mappings {
|
||||
positionsByID[mapping.CandidateID] = mapping.CandidatePosition
|
||||
}
|
||||
|
||||
groups := make([]assessedGroup, len(response.DuplicateGroups))
|
||||
owners := make(map[int][]int)
|
||||
for groupIndex, proposal := range response.DuplicateGroups {
|
||||
groups[groupIndex] = assessGroup(proposal, positionsByID)
|
||||
if !groups[groupIndex].locallyValid {
|
||||
continue
|
||||
}
|
||||
for _, position := range groups[groupIndex].memberPositions {
|
||||
owners[position] = append(owners[position], groupIndex)
|
||||
}
|
||||
}
|
||||
for _, groupIndexes := range owners {
|
||||
if len(groupIndexes) < 2 {
|
||||
continue
|
||||
}
|
||||
for _, groupIndex := range groupIndexes {
|
||||
groups[groupIndex].conflicting = true
|
||||
}
|
||||
}
|
||||
|
||||
assessment := Assessment{}
|
||||
for groupIndex, group := range groups {
|
||||
for _, category := range group.issues {
|
||||
assessment.issues = append(assessment.issues, Issue{GroupIndex: groupIndex, Category: category})
|
||||
}
|
||||
if group.conflicting {
|
||||
assessment.issues = append(assessment.issues, Issue{GroupIndex: groupIndex, Category: IssueOverlappingMember})
|
||||
}
|
||||
if !group.locallyValid || group.conflicting {
|
||||
assessment.discardedGroupCount++
|
||||
continue
|
||||
}
|
||||
assessment.plan.groups = append(assessment.plan.groups, PlanGroup{
|
||||
memberPositions: append([]int(nil), group.memberPositions...),
|
||||
canonicalPosition: group.canonicalPosition,
|
||||
})
|
||||
}
|
||||
sort.Slice(assessment.plan.groups, func(left, right int) bool {
|
||||
return assessment.plan.groups[left].memberPositions[0] < assessment.plan.groups[right].memberPositions[0]
|
||||
})
|
||||
return assessment
|
||||
}
|
||||
|
||||
func assessGroup(proposal DuplicateGroup, positionsByID map[int]int) assessedGroup {
|
||||
group := assessedGroup{}
|
||||
seenIDs := make(map[int]struct{}, len(proposal.CandidateIDs))
|
||||
memberPositions := make(map[int]struct{}, len(proposal.CandidateIDs))
|
||||
for _, candidateID := range proposal.CandidateIDs {
|
||||
if _, repeated := seenIDs[candidateID]; repeated {
|
||||
group.issues = append(group.issues, IssueRepeatedMember)
|
||||
continue
|
||||
}
|
||||
seenIDs[candidateID] = struct{}{}
|
||||
position, category := resolveMember(candidateID, positionsByID)
|
||||
if category != "" {
|
||||
group.issues = append(group.issues, category)
|
||||
continue
|
||||
}
|
||||
memberPositions[position] = struct{}{}
|
||||
group.memberPositions = append(group.memberPositions, position)
|
||||
}
|
||||
if len(memberPositions) < 2 {
|
||||
group.issues = append(group.issues, IssueFewerThanTwoMembers)
|
||||
}
|
||||
|
||||
canonicalPosition, canonicalCategory := resolveCanonical(proposal.CanonicalCandidateID, positionsByID)
|
||||
if canonicalCategory != "" {
|
||||
group.issues = append(group.issues, canonicalCategory)
|
||||
} else {
|
||||
group.canonicalPosition = canonicalPosition
|
||||
if _, member := memberPositions[canonicalPosition]; !member {
|
||||
group.issues = append(group.issues, IssueCanonicalNotMember)
|
||||
}
|
||||
}
|
||||
|
||||
sort.Ints(group.memberPositions)
|
||||
group.locallyValid = len(group.issues) == 0
|
||||
return group
|
||||
}
|
||||
|
||||
func resolveMember(candidateID int, positionsByID map[int]int) (int, IssueCategory) {
|
||||
if candidateID <= 0 {
|
||||
return 0, IssueMemberNonPositive
|
||||
}
|
||||
position, exists := positionsByID[candidateID]
|
||||
if !exists {
|
||||
return 0, IssueMemberUnknown
|
||||
}
|
||||
return position, ""
|
||||
}
|
||||
|
||||
func resolveCanonical(candidateID int, positionsByID map[int]int) (int, IssueCategory) {
|
||||
if candidateID <= 0 {
|
||||
return 0, IssueCanonicalNonPositive
|
||||
}
|
||||
position, exists := positionsByID[candidateID]
|
||||
if !exists {
|
||||
return 0, IssueCanonicalUnknown
|
||||
}
|
||||
return position, ""
|
||||
}
|
||||
|
||||
func clonePlanGroup(group PlanGroup) PlanGroup {
|
||||
return PlanGroup{
|
||||
memberPositions: append([]int(nil), group.memberPositions...),
|
||||
canonicalPosition: group.canonicalPosition,
|
||||
}
|
||||
}
|
||||
199
internal/framework/semanticreconcile/proposal_test.go
Normal file
199
internal/framework/semanticreconcile/proposal_test.go
Normal file
@@ -0,0 +1,199 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"reflect"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
)
|
||||
|
||||
func TestAssessProducesAStableOriginalPositionPlan(t *testing.T) {
|
||||
preparation := proposalPreparation(t)
|
||||
response := ProposalResponse{DuplicateGroups: []DuplicateGroup{
|
||||
{CandidateIDs: []int{5, 4}, CanonicalCandidateID: 5},
|
||||
{CandidateIDs: []int{2, 1}, CanonicalCandidateID: 2},
|
||||
}}
|
||||
assessment := preparation.Assess(response)
|
||||
want := []planGroupSnapshot{
|
||||
{members: []int{0, 2}, canonical: 2},
|
||||
{members: []int{5, 6}, canonical: 6},
|
||||
}
|
||||
if got := snapshotPlan(assessment.Plan()); !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("plan = %#v, want %#v", got, want)
|
||||
}
|
||||
if assessment.DiscardedGroupCount() != 0 || assessment.RetryRequired() || len(assessment.Issues()) != 0 {
|
||||
t.Fatalf("assessment diagnostics = discarded %d, retry %t, issues %#v", assessment.DiscardedGroupCount(), assessment.RetryRequired(), assessment.Issues())
|
||||
}
|
||||
|
||||
reordered := preparation.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{
|
||||
{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 2},
|
||||
{CandidateIDs: []int{4, 5}, CanonicalCandidateID: 5},
|
||||
}})
|
||||
if got := snapshotPlan(reordered.Plan()); !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("reordered plan = %#v, want %#v", got, want)
|
||||
}
|
||||
|
||||
empty := preparation.Assess(ProposalResponse{})
|
||||
if len(empty.Plan().Groups()) != 0 || len(empty.Issues()) != 0 || empty.DiscardedGroupCount() != 0 || empty.RetryRequired() {
|
||||
t.Fatalf("empty assessment = plan %#v, issues %#v, discarded %d, retry %t", empty.Plan().Groups(), empty.Issues(), empty.DiscardedGroupCount(), empty.RetryRequired())
|
||||
}
|
||||
}
|
||||
|
||||
func TestIssueDetailsPreservesIssueOrder(t *testing.T) {
|
||||
issues := []Issue{
|
||||
{GroupIndex: 3, Category: IssueCanonicalUnknown},
|
||||
{GroupIndex: 1, Category: IssueMemberUnknown},
|
||||
}
|
||||
want := []string{"group 3: canonical_unknown", "group 1: member_unknown"}
|
||||
if got := IssueDetails(issues); !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("IssueDetails() = %#v, want %#v", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssessRejectsEveryUnsafeLocalGroupShape(t *testing.T) {
|
||||
preparation := proposalPreparation(t)
|
||||
tests := []struct {
|
||||
name string
|
||||
group DuplicateGroup
|
||||
category IssueCategory
|
||||
}{
|
||||
{name: "zero member", group: DuplicateGroup{CandidateIDs: []int{0, 2}, CanonicalCandidateID: 2}, category: IssueMemberNonPositive},
|
||||
{name: "negative member", group: DuplicateGroup{CandidateIDs: []int{-1, 2}, CanonicalCandidateID: 2}, category: IssueMemberNonPositive},
|
||||
{name: "unknown member", group: DuplicateGroup{CandidateIDs: []int{99, 2}, CanonicalCandidateID: 2}, category: IssueMemberUnknown},
|
||||
{name: "repeated member", group: DuplicateGroup{CandidateIDs: []int{1, 1}, CanonicalCandidateID: 1}, category: IssueRepeatedMember},
|
||||
{name: "too small", group: DuplicateGroup{CandidateIDs: []int{1}, CanonicalCandidateID: 1}, category: IssueFewerThanTwoMembers},
|
||||
{name: "zero canonical", group: DuplicateGroup{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 0}, category: IssueCanonicalNonPositive},
|
||||
{name: "negative canonical", group: DuplicateGroup{CandidateIDs: []int{1, 2}, CanonicalCandidateID: -1}, category: IssueCanonicalNonPositive},
|
||||
{name: "unknown canonical", group: DuplicateGroup{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 99}, category: IssueCanonicalUnknown},
|
||||
{name: "canonical not member", group: DuplicateGroup{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 3}, category: IssueCanonicalNotMember},
|
||||
}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
assessment := preparation.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{test.group}})
|
||||
if len(assessment.Plan().Groups()) != 0 || assessment.DiscardedGroupCount() != 1 || !assessment.RetryRequired() {
|
||||
t.Fatalf("assessment = plan %#v, discarded %d, retry %t", assessment.Plan().Groups(), assessment.DiscardedGroupCount(), assessment.RetryRequired())
|
||||
}
|
||||
if !hasIssue(assessment.Issues(), 0, test.category) {
|
||||
t.Fatalf("issues = %#v, want category %q", assessment.Issues(), test.category)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssessDiscardsEveryOverlappingGroupAndRetainsIndependentGroups(t *testing.T) {
|
||||
preparation := proposalPreparation(t)
|
||||
assessment := preparation.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{
|
||||
{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 1},
|
||||
{CandidateIDs: []int{2, 3}, CanonicalCandidateID: 2},
|
||||
{CandidateIDs: []int{4, 5}, CanonicalCandidateID: 5},
|
||||
}})
|
||||
wantPlan := []planGroupSnapshot{{members: []int{5, 6}, canonical: 6}}
|
||||
if got := snapshotPlan(assessment.Plan()); !reflect.DeepEqual(got, wantPlan) {
|
||||
t.Fatalf("plan = %#v, want %#v", got, wantPlan)
|
||||
}
|
||||
if assessment.DiscardedGroupCount() != 2 || !assessment.RetryRequired() {
|
||||
t.Fatalf("discarded = %d, retry = %t", assessment.DiscardedGroupCount(), assessment.RetryRequired())
|
||||
}
|
||||
issues := assessment.Issues()
|
||||
if len(issues) != 2 || !hasIssue(issues, 0, IssueOverlappingMember) || !hasIssue(issues, 1, IssueOverlappingMember) {
|
||||
t.Fatalf("issues = %#v, want both conflicting group indexes", issues)
|
||||
}
|
||||
|
||||
invalidAndSafe := preparation.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{
|
||||
{CandidateIDs: []int{1, 99}, CanonicalCandidateID: 1},
|
||||
{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 2},
|
||||
}})
|
||||
wantPlan = []planGroupSnapshot{{members: []int{0, 2}, canonical: 2}}
|
||||
if got := snapshotPlan(invalidAndSafe.Plan()); !reflect.DeepEqual(got, wantPlan) {
|
||||
t.Fatalf("plan with independent invalid group = %#v, want %#v", got, wantPlan)
|
||||
}
|
||||
if invalidAndSafe.DiscardedGroupCount() != 1 || !invalidAndSafe.RetryRequired() || !hasIssue(invalidAndSafe.Issues(), 0, IssueMemberUnknown) {
|
||||
t.Fatalf("invalid-and-safe assessment = discarded %d, retry %t, issues %#v", invalidAndSafe.DiscardedGroupCount(), invalidAndSafe.RetryRequired(), invalidAndSafe.Issues())
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssessmentAccessorsAndInputsDoNotShareRetainedState(t *testing.T) {
|
||||
preparation := proposalPreparation(t)
|
||||
response := ProposalResponse{DuplicateGroups: []DuplicateGroup{
|
||||
{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 2},
|
||||
}}
|
||||
assessment := preparation.Assess(response)
|
||||
want := snapshotPlan(assessment.Plan())
|
||||
|
||||
response.DuplicateGroups[0].CandidateIDs[0] = 99
|
||||
response.DuplicateGroups[0].CanonicalCandidateID = 99
|
||||
plan := assessment.Plan()
|
||||
groups := plan.Groups()
|
||||
members := groups[0].MemberPositions()
|
||||
members[0] = 99
|
||||
groups[0] = PlanGroup{}
|
||||
if got := snapshotPlan(assessment.Plan()); !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("assessment plan changed through returned or input data: %#v", got)
|
||||
}
|
||||
|
||||
invalid := preparation.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{{
|
||||
CandidateIDs: []int{1, 99}, CanonicalCandidateID: 1,
|
||||
}}})
|
||||
issues := invalid.Issues()
|
||||
issues[0].GroupIndex = 99
|
||||
issues[0].Category = IssueOverlappingMember
|
||||
if retained := invalid.Issues(); retained[0].GroupIndex == 99 || retained[0].Category == IssueOverlappingMember {
|
||||
t.Fatalf("Issues() exposed retained state: %#v", retained)
|
||||
}
|
||||
}
|
||||
|
||||
type planGroupSnapshot struct {
|
||||
members []int
|
||||
canonical int
|
||||
}
|
||||
|
||||
func snapshotPlan(plan Plan) []planGroupSnapshot {
|
||||
groups := plan.Groups()
|
||||
snapshot := make([]planGroupSnapshot, len(groups))
|
||||
for index, group := range groups {
|
||||
snapshot[index] = planGroupSnapshot{
|
||||
members: group.MemberPositions(),
|
||||
canonical: group.CanonicalPosition(),
|
||||
}
|
||||
}
|
||||
return snapshot
|
||||
}
|
||||
|
||||
func hasIssue(issues []Issue, groupIndex int, category IssueCategory) bool {
|
||||
for _, issue := range issues {
|
||||
if issue.GroupIndex == groupIndex && issue.Category == category {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func proposalPreparation(t *testing.T) Preparation {
|
||||
t.Helper()
|
||||
document := &source.SourceDocument{ID: "session", Units: make([]source.SourceUnit, 7)}
|
||||
candidates := make([]Candidate, len(document.Units))
|
||||
for position := range document.Units {
|
||||
unitID := position + 1
|
||||
document.Units[position] = source.SourceUnit{ID: unitID, Text: "unit"}
|
||||
candidates[position] = Candidate{
|
||||
Label: "candidate",
|
||||
SourceRefs: []source.SourceRef{{
|
||||
SourceID: document.ID,
|
||||
StartUnitID: unitID,
|
||||
EndUnitID: unitID,
|
||||
}},
|
||||
}
|
||||
}
|
||||
for _, position := range []int{1, 4} {
|
||||
candidates[position].SourceRefs[0].SourceID = "other"
|
||||
}
|
||||
preparation, err := Prepare(document, candidates, Limits{
|
||||
ContextRadius: 0,
|
||||
MaximumCandidates: len(candidates),
|
||||
MaximumMaterialBytes: 10000,
|
||||
})
|
||||
if err != nil || preparation.Disposition() != Ready {
|
||||
t.Fatalf("Prepare() disposition = %v, error = %v", preparation.Disposition(), err)
|
||||
}
|
||||
return preparation
|
||||
}
|
||||
41
internal/framework/semanticreconcile/schema.go
Normal file
41
internal/framework/semanticreconcile/schema.go
Normal file
@@ -0,0 +1,41 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"io/fs"
|
||||
|
||||
rootassets "gitea.maximumdirect.net/eric/notarius/assets"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
)
|
||||
|
||||
const (
|
||||
ResponseSchemaKey = llm.ResponseSchemaKey("semantic_reconciliation_llm")
|
||||
ResponseSchemaID = "notarius.generic.semantic_reconciliation.llm"
|
||||
ResponseSchemaName = "notarius_semantic_reconciliation_llm_v1"
|
||||
SchemaVersion = "v1"
|
||||
SchemaAssetPath = "schemas/semantic_reconciliation_llm.v1.json"
|
||||
)
|
||||
|
||||
func assetFS() (fs.FS, error) {
|
||||
assets, err := fs.Sub(rootassets.FS(), "generic/normalize/deduplication")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("scope semantic reconciliation assets: %w", err)
|
||||
}
|
||||
return assets, nil
|
||||
}
|
||||
|
||||
// LoadResponseSchema returns the private request-local integer proposal
|
||||
// contract. It is separate from every durable artifact schema.
|
||||
func LoadResponseSchema() (llm.ResponseSchema, error) {
|
||||
assets, err := assetFS()
|
||||
if err != nil {
|
||||
return llm.ResponseSchema{}, err
|
||||
}
|
||||
return llm.LoadResponseSchema(assets, llm.ResponseSchemaDefinition{
|
||||
Key: ResponseSchemaKey,
|
||||
ID: ResponseSchemaID,
|
||||
Version: SchemaVersion,
|
||||
Name: ResponseSchemaName,
|
||||
AssetPath: SchemaAssetPath,
|
||||
})
|
||||
}
|
||||
107
internal/framework/semanticreconcile/schema_test.go
Normal file
107
internal/framework/semanticreconcile/schema_test.go
Normal file
@@ -0,0 +1,107 @@
|
||||
package semanticreconcile
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/santhosh-tekuri/jsonschema/v6"
|
||||
)
|
||||
|
||||
func TestResponseSchemaMetadataAndOwnership(t *testing.T) {
|
||||
schema, err := LoadResponseSchema()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if schema.Key != ResponseSchemaKey || schema.ID != ResponseSchemaID || schema.Version != SchemaVersion || schema.Name != ResponseSchemaName {
|
||||
t.Fatalf("schema metadata = %#v", schema)
|
||||
}
|
||||
if !json.Valid(schema.JSONSchema) || !strings.HasPrefix(schema.SHA256, "sha256:") {
|
||||
t.Fatalf("schema content metadata = %#v", schema)
|
||||
}
|
||||
|
||||
first := append([]byte(nil), schema.JSONSchema...)
|
||||
schema.JSONSchema[0] = '['
|
||||
loadedAgain, err := LoadResponseSchema()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !bytes.Equal(loadedAgain.JSONSchema, first) || !json.Valid(loadedAgain.JSONSchema) {
|
||||
t.Fatal("LoadResponseSchema() exposed shared schema content")
|
||||
}
|
||||
}
|
||||
|
||||
func TestResponseSchemaAcceptsOnlyTheIntegerProposalShape(t *testing.T) {
|
||||
schema, err := LoadResponseSchema()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
tests := []struct {
|
||||
name string
|
||||
value any
|
||||
valid bool
|
||||
}{
|
||||
{name: "empty proposal", value: map[string]any{"duplicate_groups": []any{}}, valid: true},
|
||||
{name: "valid group", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2}, "canonical_candidate_id": 1}}}, valid: true},
|
||||
{name: "semantic canonical mismatch", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2}, "canonical_candidate_id": 3}}}, valid: true},
|
||||
{name: "missing proposal", value: map[string]any{}, valid: false},
|
||||
{name: "unknown top-level field", value: map[string]any{"duplicate_groups": []any{}, "extra": true}, valid: false},
|
||||
{name: "missing members", value: map[string]any{"duplicate_groups": []any{map[string]any{"canonical_candidate_id": 1}}}, valid: false},
|
||||
{name: "missing canonical", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2}}}}, valid: false},
|
||||
{name: "unknown group field", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2}, "canonical_candidate_id": 1, "name": "replacement"}}}, valid: false},
|
||||
{name: "too few members", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1}, "canonical_candidate_id": 1}}}, valid: false},
|
||||
{name: "semantic repeated members", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 1}, "canonical_candidate_id": 1}}}, valid: true},
|
||||
{name: "zero member", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{0, 1}, "canonical_candidate_id": 1}}}, valid: false},
|
||||
{name: "negative member", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{-1, 1}, "canonical_candidate_id": 1}}}, valid: false},
|
||||
{name: "non-integer member", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2.5}, "canonical_candidate_id": 1}}}, valid: false},
|
||||
{name: "zero canonical", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2}, "canonical_candidate_id": 0}}}, valid: false},
|
||||
{name: "contextual selectors", value: map[string]any{"duplicate_groups": []any{map[string]any{"candidate_ids": []any{1, 2}, "canonical_candidate_id": 1, "source_refs": []any{}}}}, valid: false},
|
||||
}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
content, err := json.Marshal(test.value)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
err = validateAgainstSchema(content, schema.JSONSchema)
|
||||
if (err == nil) != test.valid {
|
||||
t.Fatalf("schema validation error = %v, want valid=%t", err, test.valid)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
content := []byte(`{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2}]}`)
|
||||
if err := validateAgainstSchema(content, schema.JSONSchema); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
var response ProposalResponse
|
||||
if err := json.Unmarshal(content, &response); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
want := ProposalResponse{DuplicateGroups: []DuplicateGroup{{CandidateIDs: []int{1, 2}, CanonicalCandidateID: 2}}}
|
||||
if !reflect.DeepEqual(response, want) {
|
||||
t.Fatalf("decoded response = %#v, want %#v", response, want)
|
||||
}
|
||||
}
|
||||
|
||||
func validateAgainstSchema(instanceContent, schemaContent []byte) error {
|
||||
instance, err := jsonschema.UnmarshalJSON(bytes.NewReader(instanceContent))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
document, err := jsonschema.UnmarshalJSON(bytes.NewReader(schemaContent))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
compiler := jsonschema.NewCompiler()
|
||||
if err := compiler.AddResource("schema.json", document); err != nil {
|
||||
return err
|
||||
}
|
||||
compiled, err := compiler.Compile("schema.json")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return compiled.Validate(instance)
|
||||
}
|
||||
@@ -48,13 +48,15 @@ func TestPromptAssetsPrepareItemOccurrencePrompt(t *testing.T) {
|
||||
rendered[index] = message.Content
|
||||
}
|
||||
content := strings.Join(rendered, "\n")
|
||||
for _, field := range []string{"start_unit_id", "end_unit_id", "source_id"} {
|
||||
for _, field := range []string{"start_unit_id", "end_unit_id"} {
|
||||
if !strings.Contains(content, field) {
|
||||
t.Fatalf("prepared prompt does not include shared evidence field %q", field)
|
||||
}
|
||||
}
|
||||
if strings.Contains(content, "start_segment") || strings.Contains(content, "end_segment") {
|
||||
t.Fatalf("prepared prompt contains obsolete segment evidence fields: %s", content)
|
||||
for _, obsolete := range []string{"source_id", "start_segment", "end_segment"} {
|
||||
if strings.Contains(content, obsolete) {
|
||||
t.Fatalf("prepared prompt contains obsolete evidence field %q: %s", obsolete, content)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -3,7 +3,6 @@ package itemregistry
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"reflect"
|
||||
"sort"
|
||||
@@ -13,20 +12,19 @@ import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/items/identity"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/diagnostics"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
)
|
||||
|
||||
const (
|
||||
Key = "dnd/item-registry"
|
||||
PromptID = "dnd.item_registry.normalize"
|
||||
normalizationPolicy = "dnd.item_registry.normalize.v2"
|
||||
semanticContextPolicy = "dnd.entity_reconcile.context.v1"
|
||||
semanticContextRadius = 2
|
||||
NormalizationPolicy = normalizationPolicy
|
||||
Key = "dnd/item-registry"
|
||||
PromptID = "dnd.item_registry.normalize"
|
||||
PromptVersion = "v1"
|
||||
normalizationPolicy = "dnd.item_registry.normalize.v3"
|
||||
NormalizationPolicy = normalizationPolicy
|
||||
|
||||
ReasonCodeItemFieldsNormalized = "item_fields_normalized"
|
||||
ReasonCodeItemIDRecomputed = "item_id_recomputed"
|
||||
@@ -47,9 +45,7 @@ var _ pipeline.CheckpointFingerprintProvider = (*Normalizer)(nil)
|
||||
type Options struct{}
|
||||
|
||||
type Normalizer struct {
|
||||
llm contracts.StructuredLLMClient
|
||||
promptSHA string
|
||||
responseSchemaSHA string
|
||||
engine *semanticreconcile.Engine
|
||||
}
|
||||
|
||||
func New(llmClient contracts.StructuredLLMClient, _ Options) (*Normalizer, error) {
|
||||
@@ -60,46 +56,45 @@ func New(llmClient contracts.StructuredLLMClient, _ Options) (*Normalizer, error
|
||||
if err != nil {
|
||||
return nil, normalizerErrorf("load prompt metadata: %w", err)
|
||||
}
|
||||
responseSchema, err := entityreconcile.LoadResponseSchema()
|
||||
engine, err := semanticreconcile.NewEngine(llmClient, semanticreconcile.PromptSpec{
|
||||
ID: PromptID, Version: PromptVersion, SHA256: promptSHA,
|
||||
}, semanticreconcile.DefaultLimits())
|
||||
if err != nil {
|
||||
return nil, normalizerErrorf("load response schema: %w", err)
|
||||
return nil, normalizerErrorf("construct semantic reconciliation engine: %w", err)
|
||||
}
|
||||
return &Normalizer{llm: llmClient, promptSHA: promptSHA, responseSchemaSHA: responseSchema.SHA256}, nil
|
||||
return &Normalizer{engine: engine}, nil
|
||||
}
|
||||
|
||||
func (n *Normalizer) Key() string { return Key }
|
||||
func (n *Normalizer) ReferenceSlots() []contracts.ReferenceSlot { return nil }
|
||||
|
||||
func (n *Normalizer) ManifestMetadata() map[string]any {
|
||||
if n == nil {
|
||||
if n == nil || n.engine == nil {
|
||||
return nil
|
||||
}
|
||||
return map[string]any{
|
||||
"prompt_id": PromptID, "prompt_version": entityreconcile.SchemaVersion, "prompt_sha256": n.promptSHA,
|
||||
"response_schema_key": string(entityreconcile.ResponseSchemaKey), "response_schema_id": entityreconcile.ResponseSchemaID,
|
||||
"response_schema_name": entityreconcile.ResponseSchemaName, "response_schema_version": entityreconcile.SchemaVersion,
|
||||
"response_schema_sha256": n.responseSchemaSHA, "identity_policy": identity.Policy,
|
||||
"normalization_policy": normalizationPolicy, "semantic_context_policy": semanticContextPolicy, "semantic_context_radius": semanticContextRadius,
|
||||
}
|
||||
metadata := n.engine.ManifestMetadata()
|
||||
metadata["identity_policy"] = identity.Policy
|
||||
metadata["normalization_policy"] = normalizationPolicy
|
||||
return metadata
|
||||
}
|
||||
|
||||
func (n *Normalizer) CheckpointFingerprints() []pipeline.CheckpointFingerprint {
|
||||
if n == nil {
|
||||
if n == nil || n.engine == nil {
|
||||
return nil
|
||||
}
|
||||
return []pipeline.CheckpointFingerprint{
|
||||
{Name: "prompt", Value: n.promptSHA}, {Name: "response_schema", Value: n.responseSchemaSHA},
|
||||
{Name: "identity_policy", Value: identity.Policy}, {Name: "normalization_policy", Value: normalizationPolicy},
|
||||
{Name: "semantic_context_policy", Value: fmt.Sprintf("%s:%d", semanticContextPolicy, semanticContextRadius)},
|
||||
}
|
||||
fingerprints := n.engine.CheckpointFingerprints()
|
||||
return append(fingerprints,
|
||||
pipeline.CheckpointFingerprint{Name: "identity_policy", Value: identity.Policy},
|
||||
pipeline.CheckpointFingerprint{Name: "normalization_policy", Value: normalizationPolicy},
|
||||
)
|
||||
}
|
||||
|
||||
func (n *Normalizer) Normalize(ctx context.Context, req contracts.TypedNormalizeRequest[dnd.ItemRegistry]) (contracts.TypedNormalizeResult[dnd.ItemRegistry], error) {
|
||||
if n == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("normalizer must not be nil")
|
||||
}
|
||||
if n.llm == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("LLM client must not be nil")
|
||||
if n.engine == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("semantic reconciliation engine must not be nil")
|
||||
}
|
||||
if ctx == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("context must not be nil")
|
||||
@@ -107,39 +102,51 @@ func (n *Normalizer) Normalize(ctx context.Context, req contracts.TypedNormalize
|
||||
if err := ctx.Err(); err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("context error before normalize: %w", err)
|
||||
}
|
||||
if req.Source == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("source document must not be nil")
|
||||
}
|
||||
|
||||
order := shared.NewSourceRefOrder(req.Source)
|
||||
records, warnings := preprocessRecords(req.MergeOutput.Value, order)
|
||||
deterministic := recordList(records)
|
||||
materials, ready, err := entityreconcile.BuildContext(req.Source, reconciliationCandidates(records), semanticContextRadius)
|
||||
if err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("build semantic context: %w", err)
|
||||
}
|
||||
if !ready {
|
||||
if len(records) < 2 {
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{Value: deterministic, Warnings: limitWarnings(warnings)}, nil
|
||||
}
|
||||
|
||||
var response entityreconcile.ProposalResponse
|
||||
if _, err := n.llm.CompleteStructured(ctx, contracts.StructuredCompletionRequest{
|
||||
StageName: Key, PromptID: PromptID, PromptVersion: entityreconcile.SchemaVersion,
|
||||
candidates, envelopes, err := reconciliationInputs(records)
|
||||
if err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("prepare semantic reconciliation inputs: %w", err)
|
||||
}
|
||||
reconciliation, err := n.engine.Reconcile(ctx, semanticreconcile.Request{
|
||||
StageName: Key, Source: req.Source, Candidates: candidates,
|
||||
ProfileID: req.LLMProfile, SessionID: req.SessionID,
|
||||
Inputs: contracts.LLMInputSet{"candidates": materials.Candidates, "transcript": materials.Transcript},
|
||||
}, &response); err != nil {
|
||||
if errors.Is(err, contracts.ErrInvalidStructuredOutput) {
|
||||
return n.invalidStructuredResult(deterministic, warnings), nil
|
||||
}
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("complete structured output: %w", err)
|
||||
})
|
||||
if err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("reconcile semantic duplicates: %w", err)
|
||||
}
|
||||
|
||||
switch reconciliation.Disposition() {
|
||||
case semanticreconcile.SkippedInsufficientCandidates:
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{Value: deterministic, Warnings: limitWarnings(warnings)}, nil
|
||||
case semanticreconcile.SkippedLimitExceeded:
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{Value: deterministic, Warnings: limitWarningsWithSemanticFallback(warnings)}, nil
|
||||
case semanticreconcile.RetryableInvalidStructuredOutput:
|
||||
return n.invalidStructuredResult(deterministic, warnings), nil
|
||||
case semanticreconcile.Complete, semanticreconcile.RetryableDiscardedProposalGroups:
|
||||
default:
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("unknown semantic reconciliation disposition %d", reconciliation.Disposition())
|
||||
}
|
||||
|
||||
applied, semanticWarnings, rejectedGroups, err := applyReconciliationPlan(reconciliation.Plan(), records, envelopes, order)
|
||||
if err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{}, normalizerErrorf("apply semantic reconciliation plan: %w", err)
|
||||
}
|
||||
assessment := materials.Assess(response)
|
||||
applied, semanticWarnings, rejectedGroups := applySafeGroups(records, reconciliationGroups(assessment, materials.CandidateKeys()), order)
|
||||
warnings = append(warnings, semanticWarnings...)
|
||||
if rejectedGroups > 0 {
|
||||
return currencyRetryResult(recordList(applied), warnings, rejectedGroups), nil
|
||||
}
|
||||
if assessment.DiscardedGroups() == 0 {
|
||||
discardedGroups := reconciliation.DiscardedGroupCount() + rejectedGroups
|
||||
if discardedGroups == 0 {
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{Value: recordList(applied), Warnings: limitWarnings(warnings)}, nil
|
||||
}
|
||||
return retryResult(recordList(applied), warnings, assessment), nil
|
||||
return retryResult(recordList(applied), warnings, reconciliation, rejectedGroups), nil
|
||||
}
|
||||
|
||||
func (n *Normalizer) invalidStructuredResult(value dnd.ItemRegistry, warnings []contracts.Warning) contracts.TypedNormalizeResult[dnd.ItemRegistry] {
|
||||
@@ -149,17 +156,15 @@ func (n *Normalizer) invalidStructuredResult(value dnd.ItemRegistry, warnings []
|
||||
}}
|
||||
}
|
||||
|
||||
func retryResult(value dnd.ItemRegistry, warnings []contracts.Warning, assessment entityreconcile.Assessment) contracts.TypedNormalizeResult[dnd.ItemRegistry] {
|
||||
func retryResult(value dnd.ItemRegistry, warnings []contracts.Warning, reconciliation semanticreconcile.Result, rejectedGroups int) contracts.TypedNormalizeResult[dnd.ItemRegistry] {
|
||||
details := semanticreconcile.IssueDetails(reconciliation.Issues())
|
||||
if rejectedGroups > 0 {
|
||||
details = append(details, "currency may only be consolidated with aliases of one denomination")
|
||||
}
|
||||
discardedGroups := reconciliation.DiscardedGroupCount() + rejectedGroups
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{Value: value, Warnings: limitWarningsForRetry(warnings), Retry: &contracts.NormalizeRetry{
|
||||
ReasonCode: ReasonCodeItemSemanticProposalInvalid, Message: diagnostics.Aggregate("semantic proposal requires retry", reconciliationIssues(assessment)),
|
||||
FallbackWarnings: []contracts.Warning{semanticFallbackWarning(assessment.DiscardedGroups())},
|
||||
}}
|
||||
}
|
||||
|
||||
func currencyRetryResult(value dnd.ItemRegistry, warnings []contracts.Warning, rejectedGroups int) contracts.TypedNormalizeResult[dnd.ItemRegistry] {
|
||||
return contracts.TypedNormalizeResult[dnd.ItemRegistry]{Value: value, Warnings: limitWarningsForRetry(warnings), Retry: &contracts.NormalizeRetry{
|
||||
ReasonCode: ReasonCodeItemSemanticProposalInvalid, Message: "semantic proposal requires retry: currency may only be consolidated with aliases of one denomination",
|
||||
FallbackWarnings: []contracts.Warning{semanticFallbackWarning(rejectedGroups)},
|
||||
ReasonCode: ReasonCodeItemSemanticProposalInvalid, Message: diagnostics.Aggregate("semantic proposal requires retry", details),
|
||||
FallbackWarnings: []contracts.Warning{semanticFallbackWarning(discardedGroups)},
|
||||
}}
|
||||
}
|
||||
|
||||
@@ -187,6 +192,10 @@ func limitWarningsForRetry(warnings []contracts.Warning) []contracts.Warning {
|
||||
return append(bounded, contracts.Warning{Scope: "items", ReasonCode: ReasonCodeItemNormalizationWarningsOmitted, Message: fmt.Sprintf("%d additional warning(s) omitted", len(warnings)-displayed)})
|
||||
}
|
||||
|
||||
func limitWarningsWithSemanticFallback(warnings []contracts.Warning) []contracts.Warning {
|
||||
return append(limitWarningsForRetry(warnings), semanticFallbackWarning(-1))
|
||||
}
|
||||
|
||||
type normalizedRecord struct {
|
||||
item dnd.Item
|
||||
inputIndexes []int
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"reflect"
|
||||
"strconv"
|
||||
"strings"
|
||||
@@ -14,10 +15,10 @@ import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/items/identity"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/diagnostics"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
identityvalidator "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/validate/itemregistry/identity"
|
||||
"gitea.maximumdirect.net/eric/promptkit"
|
||||
)
|
||||
@@ -35,9 +36,25 @@ func TestModuleContractAndMetadata(t *testing.T) {
|
||||
t.Fatal("New() accepted nil client")
|
||||
}
|
||||
metadata := newNormalizer(t, &recordingNormalizerClient{}).ManifestMetadata()
|
||||
if metadata["identity_policy"] != identity.Policy || metadata["response_schema_id"] != entityreconcile.ResponseSchemaID || metadata["normalization_policy"] != normalizationPolicy || metadata["semantic_context_radius"] != semanticContextRadius {
|
||||
limits, ok := metadata["semantic_reconciliation_limits"].(map[string]any)
|
||||
if !ok || metadata["identity_policy"] != identity.Policy || metadata["normalization_policy"] != normalizationPolicy || metadata["prompt_id"] != PromptID || metadata["prompt_version"] != PromptVersion || metadata["response_schema_key"] != string(semanticreconcile.ResponseSchemaKey) || metadata["response_schema_id"] != semanticreconcile.ResponseSchemaID || metadata["response_schema_name"] != semanticreconcile.ResponseSchemaName || metadata["semantic_reconciliation_policy"] != semanticreconcile.Policy || len(limits) != 3 {
|
||||
t.Fatalf("metadata = %#v", metadata)
|
||||
}
|
||||
for _, name := range []string{"prompt", "response_schema", "semantic_reconciliation_policy", "semantic_reconciliation_limits", "identity_policy", "normalization_policy"} {
|
||||
if !hasFingerprint(newNormalizer(t, &recordingNormalizerClient{}).CheckpointFingerprints(), name) {
|
||||
t.Fatalf("fingerprints missing %q", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeRejectsNilSourceDocument(t *testing.T) {
|
||||
_, err := newNormalizer(t, &recordingNormalizerClient{}).Normalize(
|
||||
context.Background(),
|
||||
contracts.TypedNormalizeRequest[dnd.ItemRegistry]{},
|
||||
)
|
||||
if err == nil || !strings.Contains(err.Error(), "source document must not be nil") {
|
||||
t.Fatalf("Normalize() error = %v, want nil source rejection", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeConsolidatesEqualNamesAcrossEvidenceWithoutMutation(t *testing.T) {
|
||||
@@ -73,10 +90,10 @@ func TestNormalizeConsolidatesEqualNamesAcrossEvidenceWithoutMutation(t *testing
|
||||
}
|
||||
var candidates struct {
|
||||
Candidates []struct {
|
||||
Name string `json:"name"`
|
||||
Label string `json:"label"`
|
||||
} `json:"candidates"`
|
||||
}
|
||||
if err := json.Unmarshal(client.requests[0].Inputs["candidates"].Content, &candidates); err != nil || len(candidates.Candidates) != 2 || candidates.Candidates[0].Name != "Rope" || candidates.Candidates[1].Name != "Lantern" {
|
||||
if err := json.Unmarshal(client.requests[0].Inputs["candidates"].Content, &candidates); err != nil || len(candidates.Candidates) != 2 || candidates.Candidates[0].Label != "Rope" || candidates.Candidates[1].Label != "Lantern" {
|
||||
t.Fatalf("semantic candidates = %#v, %v; want one candidate per comparison name", candidates, err)
|
||||
}
|
||||
repeated, repeatErr := newNormalizer(t, &recordingNormalizerClient{}).Normalize(context.Background(), normalizeRequestWithSource(result.Value, doc))
|
||||
@@ -106,7 +123,7 @@ func TestNormalizeAppliesSafeAliasProposal(t *testing.T) {
|
||||
{Name: "Compass of the Stars", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 20}}},
|
||||
{Name: "Gold Pieces", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 30, EndUnitID: 30}}},
|
||||
}}
|
||||
client := &recordingNormalizerClient{response: `{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000002"}]}`}
|
||||
client := &recordingNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2}]}`}
|
||||
result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc))
|
||||
if err != nil || result.Retry != nil || len(result.Value.Items) != 2 {
|
||||
t.Fatalf("Normalize() = %#v, %v", result, err)
|
||||
@@ -142,7 +159,7 @@ func TestNormalizeAppliesCurrencyReconciliationSafely(t *testing.T) {
|
||||
{Name: "Gold Piece", SourceRefs: ref(20)},
|
||||
{Name: "Gold Pieces", SourceRefs: ref(30)},
|
||||
},
|
||||
response: `{"duplicate_groups":[{"members":["candidate-000001","candidate-000002","candidate-000003"],"canonical":"candidate-000002"}]}`,
|
||||
response: `{"duplicate_groups":[{"candidate_ids":[1,2,3],"canonical_candidate_id":2}]}`,
|
||||
wantNames: []string{"Gold Piece"},
|
||||
wantRefCounts: []int{3},
|
||||
warning: ReasonCodeDuplicateItemCollapsed,
|
||||
@@ -153,7 +170,7 @@ func TestNormalizeAppliesCurrencyReconciliationSafely(t *testing.T) {
|
||||
{Name: "Gold Pieces", SourceRefs: ref(10)},
|
||||
{Name: "Silver Pieces", SourceRefs: ref(20)},
|
||||
},
|
||||
response: `{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000001"}]}`,
|
||||
response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":1}]}`,
|
||||
wantNames: []string{"Gold Pieces", "Silver Pieces"},
|
||||
wantRefCounts: []int{1, 1},
|
||||
wantRetry: true,
|
||||
@@ -165,7 +182,7 @@ func TestNormalizeAppliesCurrencyReconciliationSafely(t *testing.T) {
|
||||
{Name: "Gold Pieces", SourceRefs: ref(10)},
|
||||
{Name: "Longsword", SourceRefs: ref(20)},
|
||||
},
|
||||
response: `{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000001"}]}`,
|
||||
response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":1}]}`,
|
||||
wantNames: []string{"Gold Pieces", "Longsword"},
|
||||
wantRefCounts: []int{1, 1},
|
||||
wantRetry: true,
|
||||
@@ -177,7 +194,7 @@ func TestNormalizeAppliesCurrencyReconciliationSafely(t *testing.T) {
|
||||
{Name: "Star Compass", SourceRefs: ref(10)},
|
||||
{Name: "Compass of the Stars", SourceRefs: ref(20)},
|
||||
},
|
||||
response: `{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000002"}]}`,
|
||||
response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2}]}`,
|
||||
wantNames: []string{"Compass of the Stars"},
|
||||
wantRefCounts: []int{2},
|
||||
warning: ReasonCodeDuplicateItemCollapsed,
|
||||
@@ -188,7 +205,7 @@ func TestNormalizeAppliesCurrencyReconciliationSafely(t *testing.T) {
|
||||
{Name: "Gold Pieces", SourceRefs: ref(10)},
|
||||
{Name: "Longsword", SourceRefs: ref(20)},
|
||||
},
|
||||
response: `{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000002"}]}`,
|
||||
response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2}]}`,
|
||||
wantNames: []string{"Gold Pieces", "Longsword"},
|
||||
wantRefCounts: []int{1, 1},
|
||||
wantRetry: true,
|
||||
@@ -231,13 +248,67 @@ func TestNormalizePreservesCandidatesForUnsafeProposalGroups(t *testing.T) {
|
||||
{Name: "Compass", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 20}}},
|
||||
{Name: "Rope", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 30, EndUnitID: 30}}},
|
||||
}}
|
||||
client := &recordingNormalizerClient{response: `{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000001"},{"members":["candidate-000002","candidate-000003"],"canonical":"candidate-000003"},{"members":["candidate-000001","candidate-000003"],"canonical":"candidate-000099"}]}`}
|
||||
client := &recordingNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":1},{"candidate_ids":[2,3],"canonical_candidate_id":3},{"candidate_ids":[1,3],"canonical_candidate_id":99}]}`}
|
||||
result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc))
|
||||
if err != nil || result.Retry == nil || len(result.Value.Items) != 3 || !strings.Contains(result.Retry.Message, "overlapping_member") || !strings.Contains(result.Retry.Message, "canonical_unknown") || strings.Contains(result.Retry.Message, "Star Compass") || len(result.Retry.Message) > 4096 {
|
||||
t.Fatalf("Normalize() = %#v, %v; want deterministic retry fallback", result, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeAppliesIndependentGroupAndCountsAllOmissions(t *testing.T) {
|
||||
doc := semanticDocument()
|
||||
ref := func(unitID int) []source.SourceRef {
|
||||
return []source.SourceRef{{SourceID: doc.ID, StartUnitID: unitID, EndUnitID: unitID}}
|
||||
}
|
||||
input := dnd.ItemRegistry{Items: []dnd.Item{
|
||||
{Name: "Star Compass", SourceRefs: ref(10)},
|
||||
{Name: "Compass of the Stars", SourceRefs: ref(20)},
|
||||
{Name: "Gold Pieces", SourceRefs: ref(30)},
|
||||
{Name: "Silver Pieces", SourceRefs: ref(30)},
|
||||
{Name: "Rope", SourceRefs: ref(10)},
|
||||
}}
|
||||
client := &recordingNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2},{"candidate_ids":[3,4],"canonical_candidate_id":3},{"candidate_ids":[5,99],"canonical_candidate_id":5}]}`}
|
||||
result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc))
|
||||
if err != nil || result.Retry == nil {
|
||||
t.Fatalf("Normalize() = %#v, %v; want retry with independently accepted output", result, err)
|
||||
}
|
||||
wantNames := []string{"Compass of the Stars", "Gold Pieces", "Silver Pieces", "Rope"}
|
||||
if len(result.Value.Items) != len(wantNames) {
|
||||
t.Fatalf("items = %#v, want %v", result.Value.Items, wantNames)
|
||||
}
|
||||
for index, name := range wantNames {
|
||||
if result.Value.Items[index].Name != name {
|
||||
t.Fatalf("item %d = %#v, want %q", index, result.Value.Items[index], name)
|
||||
}
|
||||
}
|
||||
if !hasWarning(result.Warnings, ReasonCodeDuplicateItemCollapsed) || !hasWarning(result.Warnings, ReasonCodeItemSemanticProposalInvalid) {
|
||||
t.Fatalf("warnings = %#v, want accepted and guarded-group diagnostics", result.Warnings)
|
||||
}
|
||||
if len(result.Retry.FallbackWarnings) != 1 || !strings.Contains(result.Retry.FallbackWarnings[0].Message, "2 proposal group(s)") {
|
||||
t.Fatalf("retry = %#v, want one guarded and one malformed group counted", result.Retry)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeLimitSkipDoesNotCallLLMAndAddsBoundedFallbackWarning(t *testing.T) {
|
||||
client := &recordingNormalizerClient{}
|
||||
doc := &source.SourceDocument{ID: "session", Units: []source.SourceUnit{{ID: 1, Kind: "narration", Text: "A crowded storeroom"}}}
|
||||
limit := semanticreconcile.DefaultLimits().MaximumCandidates
|
||||
input := dnd.ItemRegistry{Items: make([]dnd.Item, limit+1)}
|
||||
for index := range input.Items {
|
||||
input.Items[index] = dnd.Item{Name: fmt.Sprintf("Item %d", index), SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 1, EndUnitID: 1}}}
|
||||
}
|
||||
result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc))
|
||||
if err != nil || result.Retry != nil {
|
||||
t.Fatalf("Normalize() = %#v, %v; want deterministic limit fallback", result, err)
|
||||
}
|
||||
if len(client.requests) != 0 || len(result.Value.Items) != limit+1 {
|
||||
t.Fatalf("completion calls = %d, items = %d; want no call and all records", len(client.requests), len(result.Value.Items))
|
||||
}
|
||||
if !hasWarning(result.Warnings, ReasonCodeItemSemanticReconciliationExhausted) || len(result.Warnings) > diagnostics.MaxWarnings {
|
||||
t.Fatalf("warnings = %#v, want bounded reconciliation fallback", result.Warnings)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeRetryFallbackErrorsWarningsAndIdempotence(t *testing.T) {
|
||||
doc := semanticDocument()
|
||||
input := dnd.ItemRegistry{Items: []dnd.Item{{Name: "Star Compass", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 10, EndUnitID: 10}}}, {Name: "Compass", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 20}}}}}
|
||||
@@ -261,7 +332,7 @@ func TestNormalizeRetryFallbackErrorsWarningsAndIdempotence(t *testing.T) {
|
||||
|
||||
func TestRegisterPromptAssetsPreparesItemNormalizationPrompt(t *testing.T) {
|
||||
registry := llm.NewAssetRegistry()
|
||||
if err := entityreconcile.RegisterSchemaAssets(registry); err != nil {
|
||||
if err := semanticreconcile.RegisterAssets(registry); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := RegisterPromptAssets(registry); err != nil {
|
||||
@@ -276,13 +347,16 @@ func TestRegisterPromptAssetsPreparesItemNormalizationPrompt(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: PromptID, PromptVersion: entityreconcile.SchemaVersion, ProfileID: "item-normalize-test", Inputs: map[string]promptkit.ArtifactRef{"candidates": promptkit.Inline(`{"candidates":[{"name":"Rope","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`), "transcript": promptkit.Inline(`{"windows":[{"units":[]}]}`)}})
|
||||
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: PromptID, PromptVersion: PromptVersion, ProfileID: "item-normalize-test", Inputs: map[string]promptkit.ArtifactRef{"candidates": promptkit.Inline(`{"candidates":[{"candidate_id":1,"label":"Rope","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`), "transcript": promptkit.Inline(`{"windows":[{"units":[]}]}`)}})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if prepared.OutputContract.SchemaPath != "dnd_entity_reconcile_llm.v1.json" || !strings.Contains(prepared.Messages[1].Content, "currency denominations") || !strings.Contains(prepared.Messages[1].Content, "materially different item types") {
|
||||
if prepared.OutputContract.SchemaPath != "semantic_reconciliation_llm.v1.json" || !strings.Contains(prepared.Messages[1].Content, "candidate_id") || !strings.Contains(prepared.Messages[1].Content, "integer") || !strings.Contains(prepared.Messages[2].Content, "currency denominations") || !strings.Contains(prepared.Messages[2].Content, "materially different item") || prepared.Messages[2].CacheControl == nil || prepared.Messages[2].CacheControl.Type != promptkit.CacheControlEphemeral {
|
||||
t.Fatalf("prepared prompt = %#v", prepared)
|
||||
}
|
||||
if prepared.Messages[4].CacheControl == nil || prepared.Messages[4].CacheControl.Type != promptkit.CacheControlEphemeral || !strings.Contains(prepared.Messages[3].Content, `"Rope"`) || strings.Contains(prepared.Messages[3].Content, `"windows"`) || !strings.Contains(prepared.Messages[4].Content, `"windows"`) || strings.Contains(prepared.Messages[4].Content, `"Rope"`) {
|
||||
t.Fatalf("prepared prompt = %#v, want isolated candidate and transcript presentation", prepared)
|
||||
}
|
||||
}
|
||||
|
||||
type recordingNormalizerClient struct {
|
||||
@@ -300,53 +374,13 @@ func (c *recordingNormalizerClient) CompleteStructured(_ context.Context, reques
|
||||
if response == "" {
|
||||
response = `{"duplicate_groups":[]}`
|
||||
}
|
||||
content, err := contextualProposalResponse(response, request.Inputs["candidates"].Content)
|
||||
if err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, err
|
||||
}
|
||||
content := []byte(response)
|
||||
if err := json.Unmarshal(content, output); err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, err
|
||||
}
|
||||
return contracts.StructuredCompletionResponse{Content: content}, nil
|
||||
}
|
||||
|
||||
func contextualProposalResponse(response string, candidateContent []byte) ([]byte, error) {
|
||||
if !strings.Contains(response, "candidate-") {
|
||||
return []byte(response), nil
|
||||
}
|
||||
var selection struct {
|
||||
DuplicateGroups []struct {
|
||||
Members []string `json:"members"`
|
||||
Canonical string `json:"canonical"`
|
||||
} `json:"duplicate_groups"`
|
||||
}
|
||||
if err := json.Unmarshal([]byte(response), &selection); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var candidates struct {
|
||||
Candidates []entityreconcile.Selector `json:"candidates"`
|
||||
}
|
||||
if err := json.Unmarshal(candidateContent, &candidates); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
selector := func(key string) entityreconcile.Selector {
|
||||
index, err := strconv.Atoi(strings.TrimPrefix(key, "candidate-"))
|
||||
if err != nil || index < 1 || index > len(candidates.Candidates) {
|
||||
return entityreconcile.Selector{Name: key, SourceRefs: []entityreconcile.SourceRange{}}
|
||||
}
|
||||
return candidates.Candidates[index-1].Clone()
|
||||
}
|
||||
proposal := entityreconcile.ProposalResponse{DuplicateGroups: make([]entityreconcile.DuplicateGroup, len(selection.DuplicateGroups))}
|
||||
for index, group := range selection.DuplicateGroups {
|
||||
members := make([]entityreconcile.Selector, len(group.Members))
|
||||
for memberIndex, key := range group.Members {
|
||||
members[memberIndex] = selector(key)
|
||||
}
|
||||
proposal.DuplicateGroups[index] = entityreconcile.DuplicateGroup{Members: members, Canonical: selector(group.Canonical)}
|
||||
}
|
||||
return json.Marshal(proposal)
|
||||
}
|
||||
|
||||
func newNormalizer(t *testing.T, client contracts.StructuredLLMClient) *Normalizer {
|
||||
t.Helper()
|
||||
normalizer, err := New(client, Options{})
|
||||
@@ -356,7 +390,10 @@ func newNormalizer(t *testing.T, client contracts.StructuredLLMClient) *Normaliz
|
||||
return normalizer
|
||||
}
|
||||
func normalizeRequest(value dnd.ItemRegistry) contracts.TypedNormalizeRequest[dnd.ItemRegistry] {
|
||||
return contracts.TypedNormalizeRequest[dnd.ItemRegistry]{MergeOutput: contracts.MergeArtifact[dnd.ItemRegistry]{Value: value}}
|
||||
return contracts.TypedNormalizeRequest[dnd.ItemRegistry]{
|
||||
Source: &source.SourceDocument{},
|
||||
MergeOutput: contracts.MergeArtifact[dnd.ItemRegistry]{Value: value},
|
||||
}
|
||||
}
|
||||
func normalizeRequestWithSource(value dnd.ItemRegistry, doc *source.SourceDocument) contracts.TypedNormalizeRequest[dnd.ItemRegistry] {
|
||||
request := normalizeRequest(value)
|
||||
@@ -374,3 +411,12 @@ func hasWarning(warnings []contracts.Warning, reason string) bool {
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func hasFingerprint(fingerprints []pipeline.CheckpointFingerprint, name string) bool {
|
||||
for _, fingerprint := range fingerprints {
|
||||
if fingerprint.Name == name && fingerprint.Value != "" {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -8,19 +8,26 @@ import (
|
||||
rootassets "gitea.maximumdirect.net/eric/notarius/assets"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/promptfs"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared"
|
||||
)
|
||||
|
||||
const promptAssetRoot = "assets/prompts"
|
||||
|
||||
var promptAssetManifest = shared.PromptAssetManifest{
|
||||
ModuleDir: PromptID,
|
||||
ModuleFiles: []promptfs.ModulePromptFile{
|
||||
{Name: "prompt.yaml", Path: "prompts/prompt.yaml"},
|
||||
{Name: "instructions.md", Path: "prompts/instructions.md"},
|
||||
{Name: "candidates.md", Path: "prompts/candidates.md"},
|
||||
},
|
||||
SharedFiles: []string{"common-dnd-system.md", "common-dnd-entity-reconciliation.md", "common-dnd-transcript-windows.md"},
|
||||
func promptAssetManifest() (shared.PromptAssetManifest, error) {
|
||||
sharedFiles, err := semanticreconcile.SharedPromptFiles()
|
||||
if err != nil {
|
||||
return shared.PromptAssetManifest{}, fmt.Errorf("load shared semantic reconciliation prompt assets: %w", err)
|
||||
}
|
||||
return shared.PromptAssetManifest{
|
||||
ModuleDir: PromptID,
|
||||
ModuleFiles: []promptfs.ModulePromptFile{
|
||||
{Name: "prompt.yaml", Path: "prompts/prompt.yaml"},
|
||||
{Name: "instructions.md", Path: "prompts/instructions.md"},
|
||||
},
|
||||
SharedFiles: []string{"common-dnd-system.md"},
|
||||
ExternalSharedFiles: sharedFiles,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func moduleAssetFS() (fs.FS, error) {
|
||||
@@ -36,7 +43,11 @@ func RegisterPromptAssets(registry *llm.AssetRegistry) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
promptFS, err := promptAssetManifest.PromptFS(assets)
|
||||
manifest, err := promptAssetManifest()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
promptFS, err := manifest.PromptFS(assets)
|
||||
if err != nil {
|
||||
return fmt.Errorf("prepare item normalization prompt assets: %w", err)
|
||||
}
|
||||
@@ -50,7 +61,12 @@ func promptAssetMetadata() (string, error) {
|
||||
promptAssetHashErr = err
|
||||
return
|
||||
}
|
||||
promptAssetHash, promptAssetHashErr = promptAssetManifest.Hash(assets)
|
||||
manifest, err := promptAssetManifest()
|
||||
if err != nil {
|
||||
promptAssetHashErr = err
|
||||
return
|
||||
}
|
||||
promptAssetHash, promptAssetHashErr = manifest.Hash(assets)
|
||||
})
|
||||
return promptAssetHash, promptAssetHashErr
|
||||
}
|
||||
|
||||
@@ -2,104 +2,106 @@ package itemregistry
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/items/identity"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/diagnostics"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
)
|
||||
|
||||
type safeReconciliationGroup struct {
|
||||
members []int
|
||||
canonical int
|
||||
}
|
||||
const incompatibleCurrencyGroup semanticreconcile.RejectionCategory = "incompatible_currency"
|
||||
|
||||
func reconciliationCandidates(records []normalizedRecord) []entityreconcile.Candidate {
|
||||
candidates := make([]entityreconcile.Candidate, len(records))
|
||||
func reconciliationInputs(records []normalizedRecord) ([]semanticreconcile.Candidate, []semanticreconcile.Record[dnd.Item], error) {
|
||||
candidates := make([]semanticreconcile.Candidate, len(records))
|
||||
envelopes := make([]semanticreconcile.Record[dnd.Item], len(records))
|
||||
for index, record := range records {
|
||||
candidates[index] = entityreconcile.Candidate{Name: record.item.Name, SourceRefs: cloneSourceRefs(record.item.SourceRefs)}
|
||||
candidates[index] = semanticreconcile.Candidate{
|
||||
Label: record.item.Name,
|
||||
SourceRefs: cloneSourceRefs(record.item.SourceRefs),
|
||||
}
|
||||
envelope, err := semanticreconcile.NewRecord(record.item, record.inputIndexes, record.earliest, cloneItem)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("record %d: %w", index, err)
|
||||
}
|
||||
envelopes[index] = envelope
|
||||
}
|
||||
return candidates
|
||||
return candidates, envelopes, nil
|
||||
}
|
||||
|
||||
func reconciliationGroups(assessment entityreconcile.Assessment, candidateKeys []string) []safeReconciliationGroup {
|
||||
positions := make(map[string]int, len(candidateKeys))
|
||||
for index, key := range candidateKeys {
|
||||
positions[key] = index
|
||||
}
|
||||
safeGroups := assessment.SafeGroups()
|
||||
groups := make([]safeReconciliationGroup, 0, len(safeGroups))
|
||||
for _, group := range safeGroups {
|
||||
members := group.Members()
|
||||
memberPositions := make([]int, len(members))
|
||||
valid := true
|
||||
for index, key := range members {
|
||||
position, ok := positions[key]
|
||||
if !ok {
|
||||
valid = false
|
||||
break
|
||||
func applyReconciliationPlan(plan semanticreconcile.Plan, records []normalizedRecord, envelopes []semanticreconcile.Record[dnd.Item], order shared.SourceRefOrder) ([]normalizedRecord, []contracts.Warning, int, error) {
|
||||
application, err := semanticreconcile.ApplyPlan(plan, envelopes, semanticreconcile.ApplicationPolicy[dnd.Item]{
|
||||
CloneValue: cloneItem,
|
||||
RejectGroup: func(members []dnd.Item, _ dnd.Item) semanticreconcile.RejectionCategory {
|
||||
if !canConsolidate(members) {
|
||||
return incompatibleCurrencyGroup
|
||||
}
|
||||
memberPositions[index] = position
|
||||
}
|
||||
canonical, ok := positions[group.Canonical()]
|
||||
if valid && ok {
|
||||
groups = append(groups, safeReconciliationGroup{members: memberPositions, canonical: canonical})
|
||||
}
|
||||
}
|
||||
return groups
|
||||
}
|
||||
|
||||
func reconciliationIssues(assessment entityreconcile.Assessment) []string {
|
||||
issues := assessment.Issues()
|
||||
details := make([]string, len(issues))
|
||||
for index, issue := range issues {
|
||||
details[index] = fmt.Sprintf("group %d: %s", issue.GroupIndex, issue.Category)
|
||||
}
|
||||
return details
|
||||
}
|
||||
|
||||
func applySafeGroups(records []normalizedRecord, groups []safeReconciliationGroup, order shared.SourceRefOrder) ([]normalizedRecord, []contracts.Warning, int) {
|
||||
byMember := make(map[int]safeReconciliationGroup, len(groups)*2)
|
||||
for _, group := range groups {
|
||||
for _, member := range group.members {
|
||||
byMember[member] = group
|
||||
}
|
||||
}
|
||||
output := make([]normalizedRecord, 0, len(records)-len(groups))
|
||||
warnings := make([]contracts.Warning, 0, len(groups))
|
||||
rejectedGroups := 0
|
||||
for index, record := range records {
|
||||
group, grouped := byMember[index]
|
||||
if !grouped {
|
||||
output = append(output, cloneRecord(record))
|
||||
continue
|
||||
}
|
||||
if group.members[0] != index {
|
||||
continue
|
||||
}
|
||||
if !canConsolidate(records, group) {
|
||||
for _, member := range group.members {
|
||||
output = append(output, cloneRecord(records[member]))
|
||||
return ""
|
||||
},
|
||||
ConsolidateGroup: func(members []dnd.Item, canonical dnd.Item) (dnd.Item, error) {
|
||||
output := cloneItem(canonical)
|
||||
output.SourceRefs = nil
|
||||
for _, member := range members {
|
||||
output.SourceRefs = append(output.SourceRefs, member.SourceRefs...)
|
||||
}
|
||||
warnings = append(warnings, contracts.Warning{Scope: itemScope(records[group.members[0]].earliest), ReasonCode: ReasonCodeItemSemanticProposalInvalid, Message: "proposal group preserved because currency may only be consolidated with aliases of one denomination"})
|
||||
rejectedGroups++
|
||||
continue
|
||||
}
|
||||
consolidated := consolidateSemanticGroup(records, group, order)
|
||||
output = append(output, consolidated)
|
||||
warnings = append(warnings, semanticDuplicateWarning(consolidated, records[group.canonical]))
|
||||
output.SourceRefs = order.Canonicalize(output.SourceRefs)
|
||||
output.ID = identity.DeriveID(output.Name)
|
||||
return output, nil
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
return nil, nil, 0, err
|
||||
}
|
||||
return output, warnings, rejectedGroups
|
||||
|
||||
applied := application.Records()
|
||||
output := make([]normalizedRecord, len(applied))
|
||||
for index, record := range applied {
|
||||
output[index] = normalizedRecord{
|
||||
item: record.Value(),
|
||||
inputIndexes: record.OriginalInputIndexes(),
|
||||
earliest: record.EarliestInputPosition(),
|
||||
}
|
||||
}
|
||||
type orderedWarning struct {
|
||||
position int
|
||||
warning contracts.Warning
|
||||
}
|
||||
orderedWarnings := make([]orderedWarning, 0, len(application.AppliedGroups())+len(application.RejectedGroups()))
|
||||
for _, event := range application.AppliedGroups() {
|
||||
provenance := event.Provenance()
|
||||
orderedWarnings = append(orderedWarnings, orderedWarning{
|
||||
position: provenance.EarliestInputPosition(),
|
||||
warning: semanticDuplicateWarning(provenance, records[provenance.CanonicalPosition()]),
|
||||
})
|
||||
}
|
||||
for _, event := range application.RejectedGroups() {
|
||||
provenance := event.Provenance()
|
||||
orderedWarnings = append(orderedWarnings, orderedWarning{
|
||||
position: provenance.EarliestInputPosition(),
|
||||
warning: contracts.Warning{
|
||||
Scope: itemScope(provenance.EarliestInputPosition()),
|
||||
ReasonCode: ReasonCodeItemSemanticProposalInvalid,
|
||||
Message: "proposal group preserved because currency may only be consolidated with aliases of one denomination",
|
||||
},
|
||||
})
|
||||
}
|
||||
sort.SliceStable(orderedWarnings, func(left, right int) bool { return orderedWarnings[left].position < orderedWarnings[right].position })
|
||||
warnings := make([]contracts.Warning, len(orderedWarnings))
|
||||
for index, entry := range orderedWarnings {
|
||||
warnings[index] = entry.warning
|
||||
}
|
||||
return output, warnings, len(application.RejectedGroups()), nil
|
||||
}
|
||||
|
||||
func canConsolidate(records []normalizedRecord, group safeReconciliationGroup) bool {
|
||||
func canConsolidate(items []dnd.Item) bool {
|
||||
denomination := ""
|
||||
hasCurrency := false
|
||||
hasNonCurrency := false
|
||||
hasConflictingDenominations := false
|
||||
for _, member := range group.members {
|
||||
current := currencyDenomination(records[member].item.Name)
|
||||
for _, item := range items {
|
||||
current := currencyDenomination(item.Name)
|
||||
if current == "" {
|
||||
hasNonCurrency = true
|
||||
continue
|
||||
@@ -131,29 +133,18 @@ func currencyDenomination(name string) string {
|
||||
}
|
||||
}
|
||||
|
||||
func consolidateSemanticGroup(records []normalizedRecord, group safeReconciliationGroup, order shared.SourceRefOrder) normalizedRecord {
|
||||
output := cloneRecord(records[group.members[0]])
|
||||
output.item.Name = records[group.canonical].item.Name
|
||||
for _, member := range group.members[1:] {
|
||||
output.item.SourceRefs = append(output.item.SourceRefs, records[member].item.SourceRefs...)
|
||||
output.inputIndexes = append(output.inputIndexes, records[member].inputIndexes...)
|
||||
if records[member].earliest < output.earliest {
|
||||
output.earliest = records[member].earliest
|
||||
}
|
||||
}
|
||||
output.inputIndexes = sortedUniqueIndexes(output.inputIndexes)
|
||||
output.item.SourceRefs = order.Canonicalize(output.item.SourceRefs)
|
||||
output.item.ID = identity.DeriveID(output.item.Name)
|
||||
return output
|
||||
}
|
||||
|
||||
func semanticDuplicateWarning(record normalizedRecord, canonical normalizedRecord) contracts.Warning {
|
||||
details := make([]string, 0, len(record.inputIndexes)+1)
|
||||
for _, inputIndex := range record.inputIndexes {
|
||||
func semanticDuplicateWarning(provenance semanticreconcile.GroupProvenance, canonical normalizedRecord) contracts.Warning {
|
||||
inputIndexes := provenance.OriginalInputIndexes()
|
||||
details := make([]string, 0, len(inputIndexes)+1)
|
||||
for _, inputIndex := range inputIndexes {
|
||||
details = append(details, fmt.Sprintf("input index %d", inputIndex))
|
||||
}
|
||||
if canonical.earliest != record.earliest {
|
||||
if canonical.earliest != provenance.EarliestInputPosition() {
|
||||
details = append(details, fmt.Sprintf("canonical display name from input index %d", canonical.earliest))
|
||||
}
|
||||
return contracts.Warning{Scope: itemScope(record.earliest), ReasonCode: ReasonCodeDuplicateItemCollapsed, Message: diagnostics.Aggregate("semantic duplicate consolidation", details)}
|
||||
return contracts.Warning{
|
||||
Scope: itemScope(provenance.EarliestInputPosition()),
|
||||
ReasonCode: ReasonCodeDuplicateItemCollapsed,
|
||||
Message: diagnostics.Aggregate("semantic duplicate consolidation", details),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,7 +4,6 @@ package locationregistry
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"reflect"
|
||||
"sort"
|
||||
@@ -14,20 +13,19 @@ import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/locations/identity"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/diagnostics"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
)
|
||||
|
||||
const (
|
||||
Key = "dnd/location-registry"
|
||||
PromptID = "dnd.location_registry.normalize"
|
||||
normalizationPolicy = "dnd.location_registry.normalize.v2"
|
||||
semanticContextPolicy = "dnd.entity_reconcile.context.v1"
|
||||
semanticContextRadius = 2
|
||||
NormalizationPolicy = normalizationPolicy
|
||||
Key = "dnd/location-registry"
|
||||
PromptID = "dnd.location_registry.normalize"
|
||||
PromptVersion = "v1"
|
||||
normalizationPolicy = "dnd.location_registry.normalize.v3"
|
||||
NormalizationPolicy = normalizationPolicy
|
||||
|
||||
ReasonCodeLocationFieldsNormalized = "location_fields_normalized"
|
||||
ReasonCodeLocationIDRecomputed = "location_id_recomputed"
|
||||
@@ -48,9 +46,7 @@ var _ pipeline.CheckpointFingerprintProvider = (*Normalizer)(nil)
|
||||
type Options struct{}
|
||||
|
||||
type Normalizer struct {
|
||||
llm contracts.StructuredLLMClient
|
||||
promptSHA string
|
||||
responseSchemaSHA string
|
||||
engine *semanticreconcile.Engine
|
||||
}
|
||||
|
||||
func New(llmClient contracts.StructuredLLMClient, _ Options) (*Normalizer, error) {
|
||||
@@ -61,46 +57,45 @@ func New(llmClient contracts.StructuredLLMClient, _ Options) (*Normalizer, error
|
||||
if err != nil {
|
||||
return nil, normalizerErrorf("load prompt metadata: %w", err)
|
||||
}
|
||||
responseSchema, err := entityreconcile.LoadResponseSchema()
|
||||
engine, err := semanticreconcile.NewEngine(llmClient, semanticreconcile.PromptSpec{
|
||||
ID: PromptID, Version: PromptVersion, SHA256: promptSHA,
|
||||
}, semanticreconcile.DefaultLimits())
|
||||
if err != nil {
|
||||
return nil, normalizerErrorf("load response schema: %w", err)
|
||||
return nil, normalizerErrorf("construct semantic reconciliation engine: %w", err)
|
||||
}
|
||||
return &Normalizer{llm: llmClient, promptSHA: promptSHA, responseSchemaSHA: responseSchema.SHA256}, nil
|
||||
return &Normalizer{engine: engine}, nil
|
||||
}
|
||||
|
||||
func (n *Normalizer) Key() string { return Key }
|
||||
func (n *Normalizer) ReferenceSlots() []contracts.ReferenceSlot { return nil }
|
||||
|
||||
func (n *Normalizer) ManifestMetadata() map[string]any {
|
||||
if n == nil {
|
||||
if n == nil || n.engine == nil {
|
||||
return nil
|
||||
}
|
||||
return map[string]any{
|
||||
"prompt_id": PromptID, "prompt_version": entityreconcile.SchemaVersion, "prompt_sha256": n.promptSHA,
|
||||
"response_schema_key": string(entityreconcile.ResponseSchemaKey), "response_schema_id": entityreconcile.ResponseSchemaID,
|
||||
"response_schema_name": entityreconcile.ResponseSchemaName, "response_schema_version": entityreconcile.SchemaVersion,
|
||||
"response_schema_sha256": n.responseSchemaSHA, "identity_policy": identity.Policy,
|
||||
"normalization_policy": normalizationPolicy, "semantic_context_policy": semanticContextPolicy, "semantic_context_radius": semanticContextRadius,
|
||||
}
|
||||
metadata := n.engine.ManifestMetadata()
|
||||
metadata["identity_policy"] = identity.Policy
|
||||
metadata["normalization_policy"] = normalizationPolicy
|
||||
return metadata
|
||||
}
|
||||
|
||||
func (n *Normalizer) CheckpointFingerprints() []pipeline.CheckpointFingerprint {
|
||||
if n == nil {
|
||||
if n == nil || n.engine == nil {
|
||||
return nil
|
||||
}
|
||||
return []pipeline.CheckpointFingerprint{
|
||||
{Name: "prompt", Value: n.promptSHA}, {Name: "response_schema", Value: n.responseSchemaSHA},
|
||||
{Name: "identity_policy", Value: identity.Policy}, {Name: "normalization_policy", Value: normalizationPolicy},
|
||||
{Name: "semantic_context_policy", Value: fmt.Sprintf("%s:%d", semanticContextPolicy, semanticContextRadius)},
|
||||
}
|
||||
fingerprints := n.engine.CheckpointFingerprints()
|
||||
return append(fingerprints,
|
||||
pipeline.CheckpointFingerprint{Name: "identity_policy", Value: identity.Policy},
|
||||
pipeline.CheckpointFingerprint{Name: "normalization_policy", Value: normalizationPolicy},
|
||||
)
|
||||
}
|
||||
|
||||
func (n *Normalizer) Normalize(ctx context.Context, req contracts.TypedNormalizeRequest[dnd.LocationRegistry]) (contracts.TypedNormalizeResult[dnd.LocationRegistry], error) {
|
||||
if n == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("normalizer must not be nil")
|
||||
}
|
||||
if n.llm == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("LLM client must not be nil")
|
||||
if n.engine == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("semantic reconciliation engine must not be nil")
|
||||
}
|
||||
if ctx == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("context must not be nil")
|
||||
@@ -108,36 +103,50 @@ func (n *Normalizer) Normalize(ctx context.Context, req contracts.TypedNormalize
|
||||
if err := ctx.Err(); err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("context error before normalize: %w", err)
|
||||
}
|
||||
if req.Source == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("source document must not be nil")
|
||||
}
|
||||
|
||||
order := shared.NewSourceRefOrder(req.Source)
|
||||
records, warnings := preprocessRecords(req.MergeOutput.Value, order)
|
||||
deterministic := recordList(records)
|
||||
materials, ready, err := entityreconcile.BuildContext(req.Source, reconciliationCandidates(records), semanticContextRadius)
|
||||
if err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("build semantic context: %w", err)
|
||||
}
|
||||
if !ready {
|
||||
if len(records) < 2 {
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{Value: deterministic, Warnings: limitWarnings(warnings)}, nil
|
||||
}
|
||||
|
||||
var response entityreconcile.ProposalResponse
|
||||
if _, err := n.llm.CompleteStructured(ctx, contracts.StructuredCompletionRequest{
|
||||
StageName: Key, PromptID: PromptID, PromptVersion: entityreconcile.SchemaVersion,
|
||||
ProfileID: req.LLMProfile, SessionID: req.SessionID,
|
||||
Inputs: contracts.LLMInputSet{"candidates": materials.Candidates, "transcript": materials.Transcript},
|
||||
}, &response); err != nil {
|
||||
if errors.Is(err, contracts.ErrInvalidStructuredOutput) {
|
||||
return n.invalidStructuredResult(deterministic, warnings), nil
|
||||
}
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("complete structured output: %w", err)
|
||||
candidates, envelopes, err := reconciliationInputs(records)
|
||||
if err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("prepare semantic reconciliation inputs: %w", err)
|
||||
}
|
||||
reconciliation, err := n.engine.Reconcile(ctx, semanticreconcile.Request{
|
||||
StageName: Key, Source: req.Source, Candidates: candidates,
|
||||
ProfileID: req.LLMProfile, SessionID: req.SessionID,
|
||||
})
|
||||
if err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("reconcile semantic duplicates: %w", err)
|
||||
}
|
||||
|
||||
switch reconciliation.Disposition() {
|
||||
case semanticreconcile.SkippedInsufficientCandidates:
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{Value: deterministic, Warnings: limitWarnings(warnings)}, nil
|
||||
case semanticreconcile.SkippedLimitExceeded:
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{Value: deterministic, Warnings: limitWarningsWithSemanticFallback(warnings)}, nil
|
||||
case semanticreconcile.RetryableInvalidStructuredOutput:
|
||||
return n.invalidStructuredResult(deterministic, warnings), nil
|
||||
case semanticreconcile.Complete, semanticreconcile.RetryableDiscardedProposalGroups:
|
||||
default:
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("unknown semantic reconciliation disposition %d", reconciliation.Disposition())
|
||||
}
|
||||
|
||||
applied, semanticWarnings, err := applyReconciliationPlan(reconciliation.Plan(), records, envelopes, order)
|
||||
if err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{}, normalizerErrorf("apply semantic reconciliation plan: %w", err)
|
||||
}
|
||||
assessment := materials.Assess(response)
|
||||
applied, semanticWarnings := applySafeGroups(records, reconciliationGroups(assessment, materials.CandidateKeys()), order)
|
||||
warnings = append(warnings, semanticWarnings...)
|
||||
if assessment.DiscardedGroups() == 0 {
|
||||
if reconciliation.Disposition() == semanticreconcile.Complete {
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{Value: recordList(applied), Warnings: limitWarnings(warnings)}, nil
|
||||
}
|
||||
return retryResult(recordList(applied), warnings, assessment), nil
|
||||
return retryResult(recordList(applied), warnings, reconciliation), nil
|
||||
}
|
||||
|
||||
func (n *Normalizer) invalidStructuredResult(value dnd.LocationRegistry, warnings []contracts.Warning) contracts.TypedNormalizeResult[dnd.LocationRegistry] {
|
||||
@@ -147,10 +156,10 @@ func (n *Normalizer) invalidStructuredResult(value dnd.LocationRegistry, warning
|
||||
}}
|
||||
}
|
||||
|
||||
func retryResult(value dnd.LocationRegistry, warnings []contracts.Warning, assessment entityreconcile.Assessment) contracts.TypedNormalizeResult[dnd.LocationRegistry] {
|
||||
func retryResult(value dnd.LocationRegistry, warnings []contracts.Warning, reconciliation semanticreconcile.Result) contracts.TypedNormalizeResult[dnd.LocationRegistry] {
|
||||
return contracts.TypedNormalizeResult[dnd.LocationRegistry]{Value: value, Warnings: limitWarningsForRetry(warnings), Retry: &contracts.NormalizeRetry{
|
||||
ReasonCode: ReasonCodeLocationSemanticProposalInvalid, Message: diagnostics.Aggregate("semantic proposal requires retry", reconciliationIssues(assessment)),
|
||||
FallbackWarnings: []contracts.Warning{semanticFallbackWarning(assessment.DiscardedGroups())},
|
||||
ReasonCode: ReasonCodeLocationSemanticProposalInvalid, Message: diagnostics.Aggregate("semantic proposal requires retry", semanticreconcile.IssueDetails(reconciliation.Issues())),
|
||||
FallbackWarnings: []contracts.Warning{semanticFallbackWarning(reconciliation.DiscardedGroupCount())},
|
||||
}}
|
||||
}
|
||||
|
||||
@@ -178,6 +187,10 @@ func limitWarningsForRetry(warnings []contracts.Warning) []contracts.Warning {
|
||||
return append(bounded, contracts.Warning{Scope: "locations", ReasonCode: ReasonCodeLocationNormalizationWarningsOmitted, Message: fmt.Sprintf("%d additional warning(s) omitted", len(warnings)-displayed)})
|
||||
}
|
||||
|
||||
func limitWarningsWithSemanticFallback(warnings []contracts.Warning) []contracts.Warning {
|
||||
return append(limitWarningsForRetry(warnings), semanticFallbackWarning(-1))
|
||||
}
|
||||
|
||||
type normalizedRecord struct {
|
||||
location dnd.Location
|
||||
inputIndexes []int
|
||||
|
||||
@@ -2,7 +2,9 @@ package locationregistry
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"reflect"
|
||||
"strconv"
|
||||
"strings"
|
||||
@@ -11,9 +13,10 @@ import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/locations/identity"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/diagnostics"
|
||||
)
|
||||
|
||||
func TestModuleContractAndMetadata(t *testing.T) {
|
||||
@@ -30,11 +33,24 @@ func TestModuleContractAndMetadata(t *testing.T) {
|
||||
}
|
||||
normalizer := newNormalizer(t, &recordingLocationNormalizerClient{})
|
||||
metadata := normalizer.ManifestMetadata()
|
||||
if metadata["identity_policy"] != identity.Policy || metadata["response_schema_id"] != entityreconcile.ResponseSchemaID || metadata["normalization_policy"] != normalizationPolicy || metadata["semantic_context_radius"] != semanticContextRadius {
|
||||
limits, ok := metadata["semantic_reconciliation_limits"].(map[string]any)
|
||||
if !ok || metadata["identity_policy"] != identity.Policy || metadata["normalization_policy"] != normalizationPolicy || metadata["prompt_id"] != PromptID || metadata["prompt_version"] != PromptVersion || metadata["response_schema_key"] != string(semanticreconcile.ResponseSchemaKey) || metadata["response_schema_id"] != semanticreconcile.ResponseSchemaID || metadata["response_schema_name"] != semanticreconcile.ResponseSchemaName || metadata["semantic_reconciliation_policy"] != semanticreconcile.Policy || len(limits) != 3 {
|
||||
t.Fatalf("metadata = %#v", metadata)
|
||||
}
|
||||
if got := normalizer.CheckpointFingerprints(); len(got) != 5 || got[2].Value != identity.Policy || got[3].Value != normalizationPolicy || got[4].Value != semanticContextPolicy+":2" {
|
||||
t.Fatalf("fingerprints = %#v", got)
|
||||
for _, name := range []string{"prompt", "response_schema", "semantic_reconciliation_policy", "semantic_reconciliation_limits", "identity_policy", "normalization_policy"} {
|
||||
if !hasFingerprint(normalizer.CheckpointFingerprints(), name) {
|
||||
t.Fatalf("fingerprints missing %q", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeRejectsNilSourceDocument(t *testing.T) {
|
||||
_, err := newNormalizer(t, &recordingLocationNormalizerClient{}).Normalize(
|
||||
context.Background(),
|
||||
contracts.TypedNormalizeRequest[dnd.LocationRegistry]{},
|
||||
)
|
||||
if err == nil || !strings.Contains(err.Error(), "source document must not be nil") {
|
||||
t.Fatalf("Normalize() error = %v, want nil source rejection", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -46,7 +62,8 @@ func TestNormalizePreparesOnlyExactDuplicatesAndRetainsSameNameAndNestedPlaces(t
|
||||
{Name: "The Tavern Cellar", SourceRefs: []source.SourceRef{{SourceID: "session", StartUnitID: 2, EndUnitID: 2}}},
|
||||
}}
|
||||
before := dnd.LocationRegistry{Locations: append([]dnd.Location(nil), input.Locations...)}
|
||||
result, err := newNormalizer(t, &recordingLocationNormalizerClient{}).Normalize(context.Background(), normalizeRequest(input))
|
||||
client := &recordingLocationNormalizerClient{}
|
||||
result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequest(input))
|
||||
if err != nil || len(result.Value.Locations) != 3 {
|
||||
t.Fatalf("Normalize() = %#v, %v; want one exact duplicate removed", result, err)
|
||||
}
|
||||
@@ -56,7 +73,7 @@ func TestNormalizePreparesOnlyExactDuplicatesAndRetainsSameNameAndNestedPlaces(t
|
||||
if got := []string{result.Value.Locations[0].Name, result.Value.Locations[1].Name, result.Value.Locations[2].Name}; !reflect.DeepEqual(got, []string{"The Tavern", "The Tavern", "The Tavern Cellar"}) {
|
||||
t.Fatalf("locations = %#v, want same names and nested place retained", got)
|
||||
}
|
||||
if result.Value.Locations[0].ID == result.Value.Locations[1].ID || !hasWarning(result.Warnings, ReasonCodeDuplicateLocationCollapsed) {
|
||||
if result.Value.Locations[0].ID == result.Value.Locations[1].ID || !hasWarning(result.Warnings, ReasonCodeDuplicateLocationCollapsed) || len(client.requests) != 0 {
|
||||
t.Fatalf("result = %#v, want evidence-anchored IDs and exact duplicate warning", result)
|
||||
}
|
||||
}
|
||||
@@ -79,7 +96,7 @@ func BenchmarkExactDuplicateGroupsManyDistinct(b *testing.B) {
|
||||
}
|
||||
|
||||
func TestNormalizeAppliesSafeAliasGroupAndUsesContextualInputs(t *testing.T) {
|
||||
client := &recordingLocationNormalizerClient{response: `{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000002"}]}`}
|
||||
client := &recordingLocationNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2}]}`}
|
||||
doc := semanticDocument()
|
||||
input := dnd.LocationRegistry{Locations: []dnd.Location{
|
||||
{Name: "Old Mill", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 10, EndUnitID: 10}}},
|
||||
@@ -107,7 +124,7 @@ func TestNormalizeRejectsUnsafeAndOverlappingGroupsWithoutLosingCandidates(t *te
|
||||
{Name: "Mill", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 20}}},
|
||||
{Name: "Tavern", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 30, EndUnitID: 30}}},
|
||||
}}
|
||||
client := &recordingLocationNormalizerClient{response: `{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000001"},{"members":["candidate-000002","candidate-000003"],"canonical":"candidate-000003"}]}`}
|
||||
client := &recordingLocationNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":1},{"candidate_ids":[2,3],"canonical_candidate_id":3}]}`}
|
||||
result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc))
|
||||
if err != nil || result.Retry == nil || len(result.Value.Locations) != 3 || !strings.Contains(result.Retry.Message, "overlapping_member") {
|
||||
t.Fatalf("Normalize() = %#v, %v; want safe retry fallback", result, err)
|
||||
@@ -117,6 +134,54 @@ func TestNormalizeRejectsUnsafeAndOverlappingGroupsWithoutLosingCandidates(t *te
|
||||
}
|
||||
}
|
||||
|
||||
func TestReconciliationCandidatesKeepSameNameEvidenceDistinct(t *testing.T) {
|
||||
doc := semanticDocument()
|
||||
records := []normalizedRecord{
|
||||
{location: dnd.Location{Name: "The Tavern", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 10, EndUnitID: 10}}}, inputIndexes: []int{0}, earliest: 0},
|
||||
{location: dnd.Location{Name: "The Tavern", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 30, EndUnitID: 30}}}, inputIndexes: []int{1}, earliest: 1},
|
||||
}
|
||||
candidates, _, err := reconciliationInputs(records)
|
||||
if err != nil {
|
||||
t.Fatalf("reconciliationInputs() error = %v", err)
|
||||
}
|
||||
preparation, err := semanticreconcile.Prepare(doc, candidates, semanticreconcile.DefaultLimits())
|
||||
if err != nil || preparation.Disposition() != semanticreconcile.Ready {
|
||||
t.Fatalf("Prepare() = %#v, %v; want ready candidates", preparation, err)
|
||||
}
|
||||
var candidateInput struct {
|
||||
Candidates []struct {
|
||||
CandidateID int `json:"candidate_id"`
|
||||
Label string `json:"label"`
|
||||
} `json:"candidates"`
|
||||
}
|
||||
if err := json.Unmarshal(preparation.Materials()["candidates"].Content, &candidateInput); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(candidateInput.Candidates) != 2 || candidateInput.Candidates[0].CandidateID != 1 || candidateInput.Candidates[1].CandidateID != 2 || candidateInput.Candidates[0].Label != "The Tavern" || candidateInput.Candidates[1].Label != "The Tavern" {
|
||||
t.Fatalf("candidate input = %#v, want distinct integer handles for equal names", candidateInput)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeLimitSkipDoesNotCallLLMAndAddsBoundedFallbackWarning(t *testing.T) {
|
||||
client := &recordingLocationNormalizerClient{}
|
||||
doc := &source.SourceDocument{ID: "session", Units: []source.SourceUnit{{ID: 1, Kind: "narration", Text: "A sprawling city"}}}
|
||||
limit := semanticreconcile.DefaultLimits().MaximumCandidates
|
||||
input := dnd.LocationRegistry{Locations: make([]dnd.Location, limit+1)}
|
||||
for index := range input.Locations {
|
||||
input.Locations[index] = dnd.Location{Name: fmt.Sprintf("Place %d", index), SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 1, EndUnitID: 1}}}
|
||||
}
|
||||
result, err := newNormalizer(t, client).Normalize(context.Background(), normalizeRequestWithSource(input, doc))
|
||||
if err != nil || result.Retry != nil {
|
||||
t.Fatalf("Normalize() = %#v, %v; want deterministic limit fallback", result, err)
|
||||
}
|
||||
if len(client.requests) != 0 || len(result.Value.Locations) != limit+1 {
|
||||
t.Fatalf("completion calls = %d, locations = %d; want no call and all records", len(client.requests), len(result.Value.Locations))
|
||||
}
|
||||
if !hasWarning(result.Warnings, ReasonCodeLocationSemanticReconciliationExhausted) || len(result.Warnings) > diagnostics.MaxWarnings {
|
||||
t.Fatalf("warnings = %#v, want bounded reconciliation fallback", result.Warnings)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeHandlesRetryFallbackAndErrors(t *testing.T) {
|
||||
doc := semanticDocument()
|
||||
input := dnd.LocationRegistry{Locations: []dnd.Location{{Name: "Old Mill", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 10, EndUnitID: 10}}}, {Name: "Mill", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 20}}}}}
|
||||
@@ -152,3 +217,12 @@ func TestNormalizeOrdersEvidenceAndIsIdempotent(t *testing.T) {
|
||||
t.Fatalf("second Normalize() = %#v, %v; want idempotent value %#v", second, err, first.Value)
|
||||
}
|
||||
}
|
||||
|
||||
func hasFingerprint(fingerprints []pipeline.CheckpointFingerprint, name string) bool {
|
||||
for _, fingerprint := range fingerprints {
|
||||
if fingerprint.Name == name && fingerprint.Value != "" {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -8,19 +8,26 @@ import (
|
||||
rootassets "gitea.maximumdirect.net/eric/notarius/assets"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/promptfs"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared"
|
||||
)
|
||||
|
||||
const promptAssetRoot = "assets/prompts"
|
||||
|
||||
var promptAssetManifest = shared.PromptAssetManifest{
|
||||
ModuleDir: PromptID,
|
||||
ModuleFiles: []promptfs.ModulePromptFile{
|
||||
{Name: "prompt.yaml", Path: "prompts/prompt.yaml"},
|
||||
{Name: "instructions.md", Path: "prompts/instructions.md"},
|
||||
{Name: "candidates.md", Path: "prompts/candidates.md"},
|
||||
},
|
||||
SharedFiles: []string{"common-dnd-system.md", "common-dnd-entity-reconciliation.md", "common-dnd-transcript-windows.md"},
|
||||
func promptAssetManifest() (shared.PromptAssetManifest, error) {
|
||||
sharedFiles, err := semanticreconcile.SharedPromptFiles()
|
||||
if err != nil {
|
||||
return shared.PromptAssetManifest{}, fmt.Errorf("load shared semantic reconciliation prompt assets: %w", err)
|
||||
}
|
||||
return shared.PromptAssetManifest{
|
||||
ModuleDir: PromptID,
|
||||
ModuleFiles: []promptfs.ModulePromptFile{
|
||||
{Name: "prompt.yaml", Path: "prompts/prompt.yaml"},
|
||||
{Name: "instructions.md", Path: "prompts/instructions.md"},
|
||||
},
|
||||
SharedFiles: []string{"common-dnd-system.md"},
|
||||
ExternalSharedFiles: sharedFiles,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func moduleAssetFS() (fs.FS, error) {
|
||||
@@ -36,7 +43,11 @@ func RegisterPromptAssets(registry *llm.AssetRegistry) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
promptFS, err := promptAssetManifest.PromptFS(assets)
|
||||
manifest, err := promptAssetManifest()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
promptFS, err := manifest.PromptFS(assets)
|
||||
if err != nil {
|
||||
return fmt.Errorf("prepare location normalization prompt assets: %w", err)
|
||||
}
|
||||
@@ -50,7 +61,12 @@ func promptAssetMetadata() (string, error) {
|
||||
promptAssetHashErr = err
|
||||
return
|
||||
}
|
||||
promptAssetHash, promptAssetHashErr = promptAssetManifest.Hash(assets)
|
||||
manifest, err := promptAssetManifest()
|
||||
if err != nil {
|
||||
promptAssetHashErr = err
|
||||
return
|
||||
}
|
||||
promptAssetHash, promptAssetHashErr = manifest.Hash(assets)
|
||||
})
|
||||
return promptAssetHash, promptAssetHashErr
|
||||
}
|
||||
|
||||
@@ -7,13 +7,13 @@ import (
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/promptkit"
|
||||
)
|
||||
|
||||
func TestRegisterPromptAssetsPreparesLocationNormalizationPrompt(t *testing.T) {
|
||||
registry := llm.NewAssetRegistry()
|
||||
if err := entityreconcile.RegisterSchemaAssets(registry); err != nil {
|
||||
if err := semanticreconcile.RegisterAssets(registry); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := RegisterPromptAssets(registry); err != nil {
|
||||
@@ -28,11 +28,11 @@ func TestRegisterPromptAssetsPreparesLocationNormalizationPrompt(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: PromptID, PromptVersion: entityreconcile.SchemaVersion, ProfileID: "location-normalize-test", Inputs: map[string]promptkit.ArtifactRef{"candidates": promptkit.Inline(`{"candidates":[{"name":"The Tavern","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`), "transcript": promptkit.Inline(`{"windows":[{"units":[]}]}`)}})
|
||||
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: PromptID, PromptVersion: PromptVersion, ProfileID: "location-normalize-test", Inputs: map[string]promptkit.ArtifactRef{"candidates": promptkit.Inline(`{"candidates":[{"candidate_id":1,"label":"The Tavern","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`), "transcript": promptkit.Inline(`{"windows":[{"units":[]}]}`)}})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if prepared.OutputContract.SchemaPath != "dnd_entity_reconcile_llm.v1.json" {
|
||||
if prepared.OutputContract.SchemaPath != "semantic_reconciliation_llm.v1.json" || !strings.Contains(prepared.Messages[1].Content, "candidate_id") || !strings.Contains(prepared.Messages[1].Content, "integer") || !strings.Contains(prepared.Messages[2].Content, "same physical place") || !strings.Contains(prepared.Messages[2].Content, "parent and child places") {
|
||||
t.Fatalf("prepared prompt = %#v", prepared)
|
||||
}
|
||||
for _, index := range []int{2, 4} {
|
||||
|
||||
@@ -4,109 +4,77 @@ import (
|
||||
"fmt"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/locations/identity"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/diagnostics"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
)
|
||||
|
||||
type safeReconciliationGroup struct {
|
||||
members []int
|
||||
canonical int
|
||||
}
|
||||
|
||||
func reconciliationCandidates(records []normalizedRecord) []entityreconcile.Candidate {
|
||||
candidates := make([]entityreconcile.Candidate, len(records))
|
||||
func reconciliationInputs(records []normalizedRecord) ([]semanticreconcile.Candidate, []semanticreconcile.Record[dnd.Location], error) {
|
||||
candidates := make([]semanticreconcile.Candidate, len(records))
|
||||
envelopes := make([]semanticreconcile.Record[dnd.Location], len(records))
|
||||
for index, record := range records {
|
||||
candidates[index] = entityreconcile.Candidate{Name: record.location.Name, SourceRefs: cloneSourceRefs(record.location.SourceRefs)}
|
||||
candidates[index] = semanticreconcile.Candidate{
|
||||
Label: record.location.Name,
|
||||
SourceRefs: cloneSourceRefs(record.location.SourceRefs),
|
||||
}
|
||||
envelope, err := semanticreconcile.NewRecord(record.location, record.inputIndexes, record.earliest, cloneLocation)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("record %d: %w", index, err)
|
||||
}
|
||||
envelopes[index] = envelope
|
||||
}
|
||||
return candidates
|
||||
return candidates, envelopes, nil
|
||||
}
|
||||
|
||||
func reconciliationGroups(assessment entityreconcile.Assessment, candidateKeys []string) []safeReconciliationGroup {
|
||||
positions := make(map[string]int, len(candidateKeys))
|
||||
for index, key := range candidateKeys {
|
||||
positions[key] = index
|
||||
}
|
||||
safeGroups := assessment.SafeGroups()
|
||||
groups := make([]safeReconciliationGroup, 0, len(safeGroups))
|
||||
for _, group := range safeGroups {
|
||||
members := group.Members()
|
||||
memberPositions := make([]int, len(members))
|
||||
valid := true
|
||||
for index, key := range members {
|
||||
position, ok := positions[key]
|
||||
if !ok {
|
||||
valid = false
|
||||
break
|
||||
func applyReconciliationPlan(plan semanticreconcile.Plan, records []normalizedRecord, envelopes []semanticreconcile.Record[dnd.Location], order shared.SourceRefOrder) ([]normalizedRecord, []contracts.Warning, error) {
|
||||
application, err := semanticreconcile.ApplyPlan(plan, envelopes, semanticreconcile.ApplicationPolicy[dnd.Location]{
|
||||
CloneValue: cloneLocation,
|
||||
ConsolidateGroup: func(members []dnd.Location, canonical dnd.Location) (dnd.Location, error) {
|
||||
output := cloneLocation(canonical)
|
||||
output.SourceRefs = nil
|
||||
for _, member := range members {
|
||||
output.SourceRefs = append(output.SourceRefs, member.SourceRefs...)
|
||||
}
|
||||
memberPositions[index] = position
|
||||
}
|
||||
canonical, ok := positions[group.Canonical()]
|
||||
if valid && ok {
|
||||
groups = append(groups, safeReconciliationGroup{members: memberPositions, canonical: canonical})
|
||||
output.SourceRefs = order.Canonicalize(output.SourceRefs)
|
||||
output.ID = identity.DeriveID(output.Name, output.SourceRefs)
|
||||
return output, nil
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
|
||||
applied := application.Records()
|
||||
output := make([]normalizedRecord, len(applied))
|
||||
for index, record := range applied {
|
||||
output[index] = normalizedRecord{
|
||||
location: record.Value(),
|
||||
inputIndexes: record.OriginalInputIndexes(),
|
||||
earliest: record.EarliestInputPosition(),
|
||||
}
|
||||
}
|
||||
return groups
|
||||
warnings := make([]contracts.Warning, 0, len(application.AppliedGroups()))
|
||||
for _, event := range application.AppliedGroups() {
|
||||
provenance := event.Provenance()
|
||||
warnings = append(warnings, semanticDuplicateWarning(provenance, records[provenance.CanonicalPosition()]))
|
||||
}
|
||||
return output, warnings, nil
|
||||
}
|
||||
|
||||
func reconciliationIssues(assessment entityreconcile.Assessment) []string {
|
||||
issues := assessment.Issues()
|
||||
details := make([]string, len(issues))
|
||||
for index, issue := range issues {
|
||||
details[index] = fmt.Sprintf("group %d: %s", issue.GroupIndex, issue.Category)
|
||||
}
|
||||
return details
|
||||
}
|
||||
|
||||
func applySafeGroups(records []normalizedRecord, groups []safeReconciliationGroup, order shared.SourceRefOrder) ([]normalizedRecord, []contracts.Warning) {
|
||||
byMember := make(map[int]safeReconciliationGroup, len(groups)*2)
|
||||
for _, group := range groups {
|
||||
for _, member := range group.members {
|
||||
byMember[member] = group
|
||||
}
|
||||
}
|
||||
output := make([]normalizedRecord, 0, len(records)-len(groups))
|
||||
warnings := make([]contracts.Warning, 0, len(groups))
|
||||
for index, record := range records {
|
||||
group, grouped := byMember[index]
|
||||
if !grouped {
|
||||
output = append(output, cloneRecord(record))
|
||||
continue
|
||||
}
|
||||
if group.members[0] != index {
|
||||
continue
|
||||
}
|
||||
consolidated := consolidateSemanticGroup(records, group, order)
|
||||
output = append(output, consolidated)
|
||||
warnings = append(warnings, semanticDuplicateWarning(consolidated, records[group.canonical]))
|
||||
}
|
||||
return output, warnings
|
||||
}
|
||||
|
||||
func consolidateSemanticGroup(records []normalizedRecord, group safeReconciliationGroup, order shared.SourceRefOrder) normalizedRecord {
|
||||
output := cloneRecord(records[group.members[0]])
|
||||
output.location.Name = records[group.canonical].location.Name
|
||||
for _, member := range group.members[1:] {
|
||||
output.location.SourceRefs = append(output.location.SourceRefs, records[member].location.SourceRefs...)
|
||||
output.inputIndexes = append(output.inputIndexes, records[member].inputIndexes...)
|
||||
if records[member].earliest < output.earliest {
|
||||
output.earliest = records[member].earliest
|
||||
}
|
||||
}
|
||||
output.inputIndexes = sortedUniqueIndexes(output.inputIndexes)
|
||||
output.location.SourceRefs = order.Canonicalize(output.location.SourceRefs)
|
||||
output.location.ID = identity.DeriveID(output.location.Name, output.location.SourceRefs)
|
||||
return output
|
||||
}
|
||||
|
||||
func semanticDuplicateWarning(record normalizedRecord, canonical normalizedRecord) contracts.Warning {
|
||||
details := make([]string, 0, len(record.inputIndexes)+1)
|
||||
for _, inputIndex := range record.inputIndexes {
|
||||
func semanticDuplicateWarning(provenance semanticreconcile.GroupProvenance, canonical normalizedRecord) contracts.Warning {
|
||||
inputIndexes := provenance.OriginalInputIndexes()
|
||||
details := make([]string, 0, len(inputIndexes)+1)
|
||||
for _, inputIndex := range inputIndexes {
|
||||
details = append(details, fmt.Sprintf("input index %d", inputIndex))
|
||||
}
|
||||
if canonical.earliest != record.earliest {
|
||||
if canonical.earliest != provenance.EarliestInputPosition() {
|
||||
details = append(details, fmt.Sprintf("canonical display name from input index %d", canonical.earliest))
|
||||
}
|
||||
return contracts.Warning{Scope: locationScope(record.earliest), ReasonCode: ReasonCodeDuplicateLocationCollapsed, Message: diagnostics.Aggregate("semantic duplicate consolidation", details)}
|
||||
return contracts.Warning{
|
||||
Scope: locationScope(provenance.EarliestInputPosition()),
|
||||
ReasonCode: ReasonCodeDuplicateLocationCollapsed,
|
||||
Message: diagnostics.Aggregate("semantic duplicate consolidation", details),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,14 +3,11 @@ package locationregistry
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
)
|
||||
|
||||
type recordingLocationNormalizerClient struct {
|
||||
@@ -28,53 +25,13 @@ func (c *recordingLocationNormalizerClient) CompleteStructured(_ context.Context
|
||||
if response == "" {
|
||||
response = `{"duplicate_groups":[]}`
|
||||
}
|
||||
content, err := contextualProposalResponse(response, request.Inputs["candidates"].Content)
|
||||
if err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, err
|
||||
}
|
||||
content := []byte(response)
|
||||
if err := json.Unmarshal(content, output); err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, err
|
||||
}
|
||||
return contracts.StructuredCompletionResponse{Content: content}, nil
|
||||
}
|
||||
|
||||
func contextualProposalResponse(response string, candidateContent []byte) ([]byte, error) {
|
||||
if !strings.Contains(response, "candidate-") {
|
||||
return []byte(response), nil
|
||||
}
|
||||
var selection struct {
|
||||
DuplicateGroups []struct {
|
||||
Members []string `json:"members"`
|
||||
Canonical string `json:"canonical"`
|
||||
} `json:"duplicate_groups"`
|
||||
}
|
||||
if err := json.Unmarshal([]byte(response), &selection); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var candidates struct {
|
||||
Candidates []entityreconcile.Selector `json:"candidates"`
|
||||
}
|
||||
if err := json.Unmarshal(candidateContent, &candidates); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
selector := func(key string) entityreconcile.Selector {
|
||||
index, err := strconv.Atoi(strings.TrimPrefix(key, "candidate-"))
|
||||
if err != nil || index < 1 || index > len(candidates.Candidates) {
|
||||
return entityreconcile.Selector{Name: key, SourceRefs: []entityreconcile.SourceRange{}}
|
||||
}
|
||||
return candidates.Candidates[index-1].Clone()
|
||||
}
|
||||
proposal := entityreconcile.ProposalResponse{DuplicateGroups: make([]entityreconcile.DuplicateGroup, len(selection.DuplicateGroups))}
|
||||
for index, group := range selection.DuplicateGroups {
|
||||
members := make([]entityreconcile.Selector, len(group.Members))
|
||||
for memberIndex, key := range group.Members {
|
||||
members[memberIndex] = selector(key)
|
||||
}
|
||||
proposal.DuplicateGroups[index] = entityreconcile.DuplicateGroup{Members: members, Canonical: selector(group.Canonical)}
|
||||
}
|
||||
return json.Marshal(proposal)
|
||||
}
|
||||
|
||||
func newNormalizer(t *testing.T, client contracts.StructuredLLMClient) *Normalizer {
|
||||
t.Helper()
|
||||
normalizer, err := New(client, Options{})
|
||||
@@ -84,7 +41,10 @@ func newNormalizer(t *testing.T, client contracts.StructuredLLMClient) *Normaliz
|
||||
return normalizer
|
||||
}
|
||||
func normalizeRequest(value dnd.LocationRegistry) contracts.TypedNormalizeRequest[dnd.LocationRegistry] {
|
||||
return contracts.TypedNormalizeRequest[dnd.LocationRegistry]{MergeOutput: contracts.MergeArtifact[dnd.LocationRegistry]{Value: value}}
|
||||
return contracts.TypedNormalizeRequest[dnd.LocationRegistry]{
|
||||
Source: &source.SourceDocument{},
|
||||
MergeOutput: contracts.MergeArtifact[dnd.LocationRegistry]{Value: value},
|
||||
}
|
||||
}
|
||||
func normalizeRequestWithSource(value dnd.LocationRegistry, doc *source.SourceDocument) contracts.TypedNormalizeRequest[dnd.LocationRegistry] {
|
||||
request := normalizeRequest(value)
|
||||
|
||||
@@ -3,7 +3,6 @@ package npcregistry
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"reflect"
|
||||
"sort"
|
||||
@@ -13,19 +12,19 @@ import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/npcs/identity"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/diagnostics"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
)
|
||||
|
||||
const (
|
||||
Key = "dnd/npc-registry"
|
||||
PromptID = "dnd.npc_registry.normalize"
|
||||
normalizationPolicy = "dnd.npc_registry.normalize.v4"
|
||||
semanticContextPolicy = "dnd.entity_reconcile.context.v1"
|
||||
NormalizationPolicy = normalizationPolicy
|
||||
Key = "dnd/npc-registry"
|
||||
PromptID = "dnd.npc_registry.normalize"
|
||||
PromptVersion = "v1"
|
||||
normalizationPolicy = "dnd.npc_registry.normalize.v5"
|
||||
NormalizationPolicy = normalizationPolicy
|
||||
|
||||
ReasonCodeNPCFieldsNormalized = "npc_fields_normalized"
|
||||
ReasonCodeNPCIDRecomputed = "npc_id_recomputed"
|
||||
@@ -46,9 +45,7 @@ var _ pipeline.CheckpointFingerprintProvider = (*Normalizer)(nil)
|
||||
type Options struct{}
|
||||
|
||||
type Normalizer struct {
|
||||
llm contracts.StructuredLLMClient
|
||||
promptSHA string
|
||||
responseSchemaSHA string
|
||||
engine *semanticreconcile.Engine
|
||||
}
|
||||
|
||||
func New(llmClient contracts.StructuredLLMClient, _ Options) (*Normalizer, error) {
|
||||
@@ -59,55 +56,45 @@ func New(llmClient contracts.StructuredLLMClient, _ Options) (*Normalizer, error
|
||||
if err != nil {
|
||||
return nil, normalizerErrorf("load prompt metadata: %w", err)
|
||||
}
|
||||
responseSchema, err := entityreconcile.LoadResponseSchema()
|
||||
engine, err := semanticreconcile.NewEngine(llmClient, semanticreconcile.PromptSpec{
|
||||
ID: PromptID, Version: PromptVersion, SHA256: promptSHA,
|
||||
}, semanticreconcile.DefaultLimits())
|
||||
if err != nil {
|
||||
return nil, normalizerErrorf("load response schema: %w", err)
|
||||
return nil, normalizerErrorf("construct semantic reconciliation engine: %w", err)
|
||||
}
|
||||
return &Normalizer{llm: llmClient, promptSHA: promptSHA, responseSchemaSHA: responseSchema.SHA256}, nil
|
||||
return &Normalizer{engine: engine}, nil
|
||||
}
|
||||
|
||||
func (n *Normalizer) Key() string { return Key }
|
||||
func (n *Normalizer) ReferenceSlots() []contracts.ReferenceSlot { return nil }
|
||||
|
||||
func (n *Normalizer) ManifestMetadata() map[string]any {
|
||||
if n == nil {
|
||||
if n == nil || n.engine == nil {
|
||||
return nil
|
||||
}
|
||||
return map[string]any{
|
||||
"prompt_id": PromptID,
|
||||
"prompt_version": entityreconcile.SchemaVersion,
|
||||
"prompt_sha256": n.promptSHA,
|
||||
"response_schema_key": string(entityreconcile.ResponseSchemaKey),
|
||||
"response_schema_id": entityreconcile.ResponseSchemaID,
|
||||
"response_schema_name": entityreconcile.ResponseSchemaName,
|
||||
"response_schema_version": entityreconcile.SchemaVersion,
|
||||
"response_schema_sha256": n.responseSchemaSHA,
|
||||
"identity_policy": identity.Policy,
|
||||
"normalization_policy": normalizationPolicy,
|
||||
"semantic_context_policy": semanticContextPolicy,
|
||||
"semantic_context_radius": semanticContextRadius,
|
||||
}
|
||||
metadata := n.engine.ManifestMetadata()
|
||||
metadata["identity_policy"] = identity.Policy
|
||||
metadata["normalization_policy"] = normalizationPolicy
|
||||
return metadata
|
||||
}
|
||||
|
||||
func (n *Normalizer) CheckpointFingerprints() []pipeline.CheckpointFingerprint {
|
||||
if n == nil {
|
||||
if n == nil || n.engine == nil {
|
||||
return nil
|
||||
}
|
||||
return []pipeline.CheckpointFingerprint{
|
||||
{Name: "prompt", Value: n.promptSHA},
|
||||
{Name: "response_schema", Value: n.responseSchemaSHA},
|
||||
{Name: "identity_policy", Value: identity.Policy},
|
||||
{Name: "normalization_policy", Value: normalizationPolicy},
|
||||
{Name: "semantic_context_policy", Value: fmt.Sprintf("%s:%d", semanticContextPolicy, semanticContextRadius)},
|
||||
}
|
||||
fingerprints := n.engine.CheckpointFingerprints()
|
||||
return append(fingerprints,
|
||||
pipeline.CheckpointFingerprint{Name: "identity_policy", Value: identity.Policy},
|
||||
pipeline.CheckpointFingerprint{Name: "normalization_policy", Value: normalizationPolicy},
|
||||
)
|
||||
}
|
||||
|
||||
func (n *Normalizer) Normalize(ctx context.Context, req contracts.TypedNormalizeRequest[dnd.NPCRegistry]) (contracts.TypedNormalizeResult[dnd.NPCRegistry], error) {
|
||||
if n == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("normalizer must not be nil")
|
||||
}
|
||||
if n.llm == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("LLM client must not be nil")
|
||||
if n.engine == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("semantic reconciliation engine must not be nil")
|
||||
}
|
||||
if ctx == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("context must not be nil")
|
||||
@@ -115,37 +102,50 @@ func (n *Normalizer) Normalize(ctx context.Context, req contracts.TypedNormalize
|
||||
if err := ctx.Err(); err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("context error before normalize: %w", err)
|
||||
}
|
||||
if req.Source == nil {
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("source document must not be nil")
|
||||
}
|
||||
|
||||
order := shared.NewSourceRefOrder(req.Source)
|
||||
records, warnings := preprocessRecords(req.MergeOutput.Value, order)
|
||||
deterministic := recordList(records)
|
||||
materials, ready, err := entityreconcile.BuildContext(req.Source, reconciliationCandidates(records), semanticContextRadius)
|
||||
if err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("build semantic context: %w", err)
|
||||
}
|
||||
if !ready {
|
||||
if len(records) < 2 {
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{Value: deterministic, Warnings: limitWarnings(warnings)}, nil
|
||||
}
|
||||
|
||||
var response entityreconcile.ProposalResponse
|
||||
if _, err := n.llm.CompleteStructured(ctx, contracts.StructuredCompletionRequest{
|
||||
StageName: Key, PromptID: PromptID, PromptVersion: entityreconcile.SchemaVersion,
|
||||
candidates, envelopes, err := reconciliationInputs(records)
|
||||
if err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("prepare semantic reconciliation inputs: %w", err)
|
||||
}
|
||||
reconciliation, err := n.engine.Reconcile(ctx, semanticreconcile.Request{
|
||||
StageName: Key, Source: req.Source, Candidates: candidates,
|
||||
ProfileID: req.LLMProfile, SessionID: req.SessionID,
|
||||
Inputs: contracts.LLMInputSet{"candidates": materials.Candidates, "transcript": materials.Transcript},
|
||||
}, &response); err != nil {
|
||||
if errors.Is(err, contracts.ErrInvalidStructuredOutput) {
|
||||
return n.invalidStructuredResult(deterministic, warnings), nil
|
||||
}
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("complete structured output: %w", err)
|
||||
})
|
||||
if err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("reconcile semantic duplicates: %w", err)
|
||||
}
|
||||
|
||||
assessment := materials.Assess(response)
|
||||
applied, semanticWarnings := applySafeGroups(records, reconciliationGroups(assessment, materials.CandidateKeys()), order)
|
||||
switch reconciliation.Disposition() {
|
||||
case semanticreconcile.SkippedInsufficientCandidates:
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{Value: deterministic, Warnings: limitWarnings(warnings)}, nil
|
||||
case semanticreconcile.SkippedLimitExceeded:
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{Value: deterministic, Warnings: limitWarningsWithSemanticFallback(warnings)}, nil
|
||||
case semanticreconcile.RetryableInvalidStructuredOutput:
|
||||
return n.invalidStructuredResult(deterministic, warnings), nil
|
||||
case semanticreconcile.Complete, semanticreconcile.RetryableDiscardedProposalGroups:
|
||||
default:
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("unknown semantic reconciliation disposition %d", reconciliation.Disposition())
|
||||
}
|
||||
|
||||
applied, semanticWarnings, err := applyReconciliationPlan(reconciliation.Plan(), records, envelopes, order)
|
||||
if err != nil {
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{}, normalizerErrorf("apply semantic reconciliation plan: %w", err)
|
||||
}
|
||||
warnings = append(warnings, semanticWarnings...)
|
||||
if assessment.DiscardedGroups() == 0 {
|
||||
if reconciliation.Disposition() == semanticreconcile.Complete {
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{Value: recordList(applied), Warnings: limitWarnings(warnings)}, nil
|
||||
}
|
||||
return retryResult(recordList(applied), warnings, assessment), nil
|
||||
return retryResult(recordList(applied), warnings, reconciliation), nil
|
||||
}
|
||||
|
||||
func (n *Normalizer) invalidStructuredResult(value dnd.NPCRegistry, warnings []contracts.Warning) contracts.TypedNormalizeResult[dnd.NPCRegistry] {
|
||||
@@ -160,14 +160,14 @@ func (n *Normalizer) invalidStructuredResult(value dnd.NPCRegistry, warnings []c
|
||||
}
|
||||
}
|
||||
|
||||
func retryResult(value dnd.NPCRegistry, warnings []contracts.Warning, assessment entityreconcile.Assessment) contracts.TypedNormalizeResult[dnd.NPCRegistry] {
|
||||
func retryResult(value dnd.NPCRegistry, warnings []contracts.Warning, reconciliation semanticreconcile.Result) contracts.TypedNormalizeResult[dnd.NPCRegistry] {
|
||||
return contracts.TypedNormalizeResult[dnd.NPCRegistry]{
|
||||
Value: value,
|
||||
Warnings: limitWarningsForRetry(warnings),
|
||||
Retry: &contracts.NormalizeRetry{
|
||||
ReasonCode: ReasonCodeNPCSemanticProposalInvalid,
|
||||
Message: diagnostics.Aggregate("semantic proposal requires retry", reconciliationIssues(assessment)),
|
||||
FallbackWarnings: []contracts.Warning{semanticFallbackWarning(assessment.DiscardedGroups())},
|
||||
Message: diagnostics.Aggregate("semantic proposal requires retry", semanticreconcile.IssueDetails(reconciliation.Issues())),
|
||||
FallbackWarnings: []contracts.Warning{semanticFallbackWarning(reconciliation.DiscardedGroupCount())},
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -199,6 +199,10 @@ func limitWarningsForRetry(warnings []contracts.Warning) []contracts.Warning {
|
||||
})
|
||||
}
|
||||
|
||||
func limitWarningsWithSemanticFallback(warnings []contracts.Warning) []contracts.Warning {
|
||||
return append(limitWarningsForRetry(warnings), semanticFallbackWarning(-1))
|
||||
}
|
||||
|
||||
type normalizedRecord struct {
|
||||
npc dnd.NPC
|
||||
inputIndexes []int
|
||||
|
||||
@@ -4,16 +4,15 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"reflect"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/npcs/identity"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
)
|
||||
|
||||
func TestModuleContractAndIdentity(t *testing.T) {
|
||||
@@ -35,12 +34,16 @@ func TestModuleContractAndIdentity(t *testing.T) {
|
||||
t.Fatal("New(nil, Options{}) error = nil, want nil client rejection")
|
||||
}
|
||||
normalizer := newNormalizer(t, &recordingNPCNormalizerClient{})
|
||||
if metadata := normalizer.ManifestMetadata(); metadata["identity_policy"] != identity.Policy || metadata["normalization_policy"] != normalizationPolicy || metadata["prompt_id"] != PromptID || metadata["response_schema_key"] != string(entityreconcile.ResponseSchemaKey) || metadata["response_schema_id"] != entityreconcile.ResponseSchemaID || metadata["response_schema_name"] != entityreconcile.ResponseSchemaName || metadata["semantic_context_policy"] != semanticContextPolicy || metadata["semantic_context_radius"] != semanticContextRadius {
|
||||
metadata := normalizer.ManifestMetadata()
|
||||
limits, ok := metadata["semantic_reconciliation_limits"].(map[string]any)
|
||||
if !ok || metadata["identity_policy"] != identity.Policy || metadata["normalization_policy"] != normalizationPolicy || metadata["prompt_id"] != PromptID || metadata["prompt_version"] != PromptVersion || metadata["response_schema_key"] != string(semanticreconcile.ResponseSchemaKey) || metadata["response_schema_id"] != semanticreconcile.ResponseSchemaID || metadata["response_schema_name"] != semanticreconcile.ResponseSchemaName || metadata["semantic_reconciliation_policy"] != semanticreconcile.Policy || len(limits) != 3 {
|
||||
t.Fatalf("metadata = %#v", metadata)
|
||||
}
|
||||
wantFingerprints := []pipeline.CheckpointFingerprint{{Name: "prompt", Value: normalizer.promptSHA}, {Name: "response_schema", Value: normalizer.responseSchemaSHA}, {Name: "identity_policy", Value: identity.Policy}, {Name: "normalization_policy", Value: normalizationPolicy}, {Name: "semantic_context_policy", Value: semanticContextPolicy + ":2"}}
|
||||
if got := normalizer.CheckpointFingerprints(); !reflect.DeepEqual(got, wantFingerprints) {
|
||||
t.Fatalf("fingerprints = %#v, want %#v", got, wantFingerprints)
|
||||
fingerprints := normalizer.CheckpointFingerprints()
|
||||
for _, name := range []string{"prompt", "response_schema", "semantic_reconciliation_policy", "semantic_reconciliation_limits", "identity_policy", "normalization_policy"} {
|
||||
if !hasFingerprint(fingerprints, name) {
|
||||
t.Fatalf("fingerprints = %#v, want %q", fingerprints, name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -123,6 +126,9 @@ func TestNormalizePreservesInvalidCandidatesAndDoesNotAliasInput(t *testing.T) {
|
||||
|
||||
func TestNormalizeHandlesNilAndCanceledCalls(t *testing.T) {
|
||||
normalizer := newNormalizer(t, &recordingNPCNormalizerClient{})
|
||||
if _, err := normalizer.Normalize(context.Background(), contracts.TypedNormalizeRequest[dnd.NPCRegistry]{}); err == nil || !strings.Contains(err.Error(), "source document must not be nil") {
|
||||
t.Fatalf("nil source Normalize() error = %v", err)
|
||||
}
|
||||
result, err := normalizer.Normalize(context.Background(), normalizeRequest(dnd.NPCRegistry{NPCs: nil}))
|
||||
if err != nil || result.Value.NPCs != nil {
|
||||
t.Fatalf("nil list result = %#v, error = %v", result.Value, err)
|
||||
@@ -157,53 +163,13 @@ func (c *recordingNPCNormalizerClient) CompleteStructured(_ context.Context, req
|
||||
if response == "" {
|
||||
response = `{"duplicate_groups":[]}`
|
||||
}
|
||||
content, err := contextualProposalResponse(response, request.Inputs["candidates"].Content)
|
||||
if err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, err
|
||||
}
|
||||
content := []byte(response)
|
||||
if err := json.Unmarshal(content, output); err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, err
|
||||
}
|
||||
return contracts.StructuredCompletionResponse{Content: content}, nil
|
||||
}
|
||||
|
||||
func contextualProposalResponse(response string, candidateContent []byte) ([]byte, error) {
|
||||
if !strings.Contains(response, "candidate-") {
|
||||
return []byte(response), nil
|
||||
}
|
||||
var selection struct {
|
||||
DuplicateGroups []struct {
|
||||
Members []string `json:"members"`
|
||||
Canonical string `json:"canonical"`
|
||||
} `json:"duplicate_groups"`
|
||||
}
|
||||
if err := json.Unmarshal([]byte(response), &selection); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var candidates struct {
|
||||
Candidates []entityreconcile.Selector `json:"candidates"`
|
||||
}
|
||||
if err := json.Unmarshal(candidateContent, &candidates); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
selector := func(key string) entityreconcile.Selector {
|
||||
index, err := strconv.Atoi(strings.TrimPrefix(key, "candidate-"))
|
||||
if err != nil || index < 1 || index > len(candidates.Candidates) {
|
||||
return entityreconcile.Selector{Name: key, SourceRefs: []entityreconcile.SourceRange{}}
|
||||
}
|
||||
return candidates.Candidates[index-1].Clone()
|
||||
}
|
||||
proposal := entityreconcile.ProposalResponse{DuplicateGroups: make([]entityreconcile.DuplicateGroup, len(selection.DuplicateGroups))}
|
||||
for index, group := range selection.DuplicateGroups {
|
||||
members := make([]entityreconcile.Selector, len(group.Members))
|
||||
for memberIndex, key := range group.Members {
|
||||
members[memberIndex] = selector(key)
|
||||
}
|
||||
proposal.DuplicateGroups[index] = entityreconcile.DuplicateGroup{Members: members, Canonical: selector(group.Canonical)}
|
||||
}
|
||||
return json.Marshal(proposal)
|
||||
}
|
||||
|
||||
func newNormalizer(t *testing.T, client contracts.StructuredLLMClient) *Normalizer {
|
||||
t.Helper()
|
||||
normalizer, err := New(client, Options{})
|
||||
@@ -214,7 +180,10 @@ func newNormalizer(t *testing.T, client contracts.StructuredLLMClient) *Normaliz
|
||||
}
|
||||
|
||||
func normalizeRequest(value dnd.NPCRegistry) contracts.TypedNormalizeRequest[dnd.NPCRegistry] {
|
||||
return contracts.TypedNormalizeRequest[dnd.NPCRegistry]{MergeOutput: contracts.MergeArtifact[dnd.NPCRegistry]{Value: value}}
|
||||
return contracts.TypedNormalizeRequest[dnd.NPCRegistry]{
|
||||
Source: &source.SourceDocument{},
|
||||
MergeOutput: contracts.MergeArtifact[dnd.NPCRegistry]{Value: value},
|
||||
}
|
||||
}
|
||||
|
||||
func normalizeRequestWithSource(value dnd.NPCRegistry, doc *source.SourceDocument) contracts.TypedNormalizeRequest[dnd.NPCRegistry] {
|
||||
@@ -231,3 +200,12 @@ func hasWarning(warnings []contracts.Warning, reason, scope string) bool {
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func hasFingerprint(fingerprints []pipeline.CheckpointFingerprint, name string) bool {
|
||||
for _, fingerprint := range fingerprints {
|
||||
if fingerprint.Name == name && fingerprint.Value != "" {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -8,23 +8,26 @@ import (
|
||||
rootassets "gitea.maximumdirect.net/eric/notarius/assets"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/promptfs"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared"
|
||||
)
|
||||
|
||||
const promptAssetRoot = "assets/prompts"
|
||||
|
||||
var promptAssetManifest = shared.PromptAssetManifest{
|
||||
ModuleDir: PromptID,
|
||||
ModuleFiles: []promptfs.ModulePromptFile{
|
||||
{Name: "prompt.yaml", Path: "prompts/prompt.yaml"},
|
||||
{Name: "instructions.md", Path: "prompts/instructions.md"},
|
||||
{Name: "candidates.md", Path: "prompts/candidates.md"},
|
||||
},
|
||||
SharedFiles: []string{
|
||||
"common-dnd-system.md",
|
||||
"common-dnd-entity-reconciliation.md",
|
||||
"common-dnd-transcript-windows.md",
|
||||
},
|
||||
func promptAssetManifest() (shared.PromptAssetManifest, error) {
|
||||
sharedFiles, err := semanticreconcile.SharedPromptFiles()
|
||||
if err != nil {
|
||||
return shared.PromptAssetManifest{}, fmt.Errorf("load shared semantic reconciliation prompt assets: %w", err)
|
||||
}
|
||||
return shared.PromptAssetManifest{
|
||||
ModuleDir: PromptID,
|
||||
ModuleFiles: []promptfs.ModulePromptFile{
|
||||
{Name: "prompt.yaml", Path: "prompts/prompt.yaml"},
|
||||
{Name: "instructions.md", Path: "prompts/instructions.md"},
|
||||
},
|
||||
SharedFiles: []string{"common-dnd-system.md"},
|
||||
ExternalSharedFiles: sharedFiles,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func moduleAssetFS() (fs.FS, error) {
|
||||
@@ -40,7 +43,11 @@ func RegisterPromptAssets(registry *llm.AssetRegistry) error {
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
promptFS, err := promptAssetManifest.PromptFS(assets)
|
||||
manifest, err := promptAssetManifest()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
promptFS, err := manifest.PromptFS(assets)
|
||||
if err != nil {
|
||||
return fmt.Errorf("prepare NPC normalization prompt assets: %w", err)
|
||||
}
|
||||
@@ -54,7 +61,12 @@ func promptAssetMetadata() (string, error) {
|
||||
promptAssetHashErr = err
|
||||
return
|
||||
}
|
||||
promptAssetHash, promptAssetHashErr = promptAssetManifest.Hash(assets)
|
||||
manifest, err := promptAssetManifest()
|
||||
if err != nil {
|
||||
promptAssetHashErr = err
|
||||
return
|
||||
}
|
||||
promptAssetHash, promptAssetHashErr = manifest.Hash(assets)
|
||||
})
|
||||
return promptAssetHash, promptAssetHashErr
|
||||
}
|
||||
|
||||
@@ -7,7 +7,7 @@ import (
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/promptkit"
|
||||
)
|
||||
|
||||
@@ -16,8 +16,8 @@ func TestRegisterPromptAssetsPreparesNormalizationPrompt(t *testing.T) {
|
||||
t.Fatalf("promptAssetMetadata() = %q, %v; want prompt fingerprint", promptHash, err)
|
||||
}
|
||||
registry := llm.NewAssetRegistry()
|
||||
if err := entityreconcile.RegisterSchemaAssets(registry); err != nil {
|
||||
t.Fatalf("RegisterSchemaAssets() error = %v", err)
|
||||
if err := semanticreconcile.RegisterAssets(registry); err != nil {
|
||||
t.Fatalf("RegisterAssets() error = %v", err)
|
||||
}
|
||||
if err := RegisterPromptAssets(registry); err != nil {
|
||||
t.Fatalf("RegisterPromptAssets() error = %v", err)
|
||||
@@ -34,21 +34,27 @@ func TestRegisterPromptAssetsPreparesNormalizationPrompt(t *testing.T) {
|
||||
t.Fatalf("NewEngine() error = %v", err)
|
||||
}
|
||||
prepared, err := engine.Prepare(context.Background(), promptkit.RunRequest{
|
||||
PromptID: PromptID, PromptVersion: entityreconcile.SchemaVersion, ProfileID: "normalize-test-profile",
|
||||
PromptID: PromptID, PromptVersion: PromptVersion, ProfileID: "normalize-test-profile",
|
||||
Inputs: map[string]promptkit.ArtifactRef{
|
||||
"candidates": promptkit.Inline(`{"candidates":[{"name":"Mira","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`),
|
||||
"candidates": promptkit.Inline(`{"candidates":[{"candidate_id":1,"label":"Mira","source_refs":[{"start_unit_id":1,"end_unit_id":1}]}]}`),
|
||||
"transcript": promptkit.Inline(`{"windows":[{"units":[]}]}`),
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Prepare() error = %v", err)
|
||||
}
|
||||
if prepared.PromptID != PromptID || prepared.OutputContract.SchemaPath != "dnd_entity_reconcile_llm.v1.json" {
|
||||
if prepared.PromptID != PromptID || prepared.OutputContract.SchemaPath != "semantic_reconciliation_llm.v1.json" {
|
||||
t.Fatalf("prepared prompt = %#v, want normalization prompt identity and schema", prepared)
|
||||
}
|
||||
if prepared.Messages[0].Role != "system" {
|
||||
t.Fatalf("initial message role = %q, want system", prepared.Messages[0].Role)
|
||||
}
|
||||
if !strings.Contains(prepared.Messages[1].Content, "candidate_id") || !strings.Contains(prepared.Messages[1].Content, "integer") {
|
||||
t.Fatalf("protocol message = %q, want shared integer-handle protocol", prepared.Messages[1].Content)
|
||||
}
|
||||
if !strings.Contains(prepared.Messages[2].Content, "same individual") || strings.Contains(prepared.Messages[2].Content, "source ranges") {
|
||||
t.Fatalf("NPC policy message = %q, want domain distinctions without copied ranges", prepared.Messages[2].Content)
|
||||
}
|
||||
for _, index := range []int{2, 4} {
|
||||
if cache := prepared.Messages[index].CacheControl; cache == nil || cache.Type != promptkit.CacheControlEphemeral {
|
||||
t.Errorf("message %d cache control = %#v, want ephemeral", index, cache)
|
||||
|
||||
@@ -4,118 +4,76 @@ import (
|
||||
"fmt"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/npcs/identity"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/diagnostics"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
)
|
||||
|
||||
const semanticContextRadius = 2
|
||||
|
||||
type safeReconciliationGroup struct {
|
||||
members []int
|
||||
canonical int
|
||||
}
|
||||
|
||||
func reconciliationCandidates(records []normalizedRecord) []entityreconcile.Candidate {
|
||||
candidates := make([]entityreconcile.Candidate, len(records))
|
||||
func reconciliationInputs(records []normalizedRecord) ([]semanticreconcile.Candidate, []semanticreconcile.Record[dnd.NPC], error) {
|
||||
candidates := make([]semanticreconcile.Candidate, len(records))
|
||||
envelopes := make([]semanticreconcile.Record[dnd.NPC], len(records))
|
||||
for index, record := range records {
|
||||
candidates[index] = entityreconcile.Candidate{
|
||||
Name: record.npc.Name,
|
||||
candidates[index] = semanticreconcile.Candidate{
|
||||
Label: record.npc.Name,
|
||||
SourceRefs: cloneSourceRefs(record.npc.SourceRefs),
|
||||
}
|
||||
envelope, err := semanticreconcile.NewRecord(record.npc, record.inputIndexes, record.earliest, cloneNPC)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("record %d: %w", index, err)
|
||||
}
|
||||
envelopes[index] = envelope
|
||||
}
|
||||
return candidates
|
||||
return candidates, envelopes, nil
|
||||
}
|
||||
|
||||
func reconciliationGroups(assessment entityreconcile.Assessment, candidateKeys []string) []safeReconciliationGroup {
|
||||
positions := make(map[string]int, len(candidateKeys))
|
||||
for index, key := range candidateKeys {
|
||||
positions[key] = index
|
||||
}
|
||||
safeGroups := assessment.SafeGroups()
|
||||
groups := make([]safeReconciliationGroup, 0, len(safeGroups))
|
||||
for _, group := range safeGroups {
|
||||
members := group.Members()
|
||||
memberPositions := make([]int, len(members))
|
||||
valid := true
|
||||
for index, key := range members {
|
||||
position, ok := positions[key]
|
||||
if !ok {
|
||||
valid = false
|
||||
break
|
||||
func applyReconciliationPlan(plan semanticreconcile.Plan, records []normalizedRecord, envelopes []semanticreconcile.Record[dnd.NPC], order shared.SourceRefOrder) ([]normalizedRecord, []contracts.Warning, error) {
|
||||
application, err := semanticreconcile.ApplyPlan(plan, envelopes, semanticreconcile.ApplicationPolicy[dnd.NPC]{
|
||||
CloneValue: cloneNPC,
|
||||
ConsolidateGroup: func(members []dnd.NPC, canonical dnd.NPC) (dnd.NPC, error) {
|
||||
output := cloneNPC(canonical)
|
||||
output.SourceRefs = nil
|
||||
for _, member := range members {
|
||||
output.SourceRefs = append(output.SourceRefs, member.SourceRefs...)
|
||||
}
|
||||
memberPositions[index] = position
|
||||
}
|
||||
canonical, ok := positions[group.Canonical()]
|
||||
if !valid || !ok {
|
||||
continue
|
||||
}
|
||||
groups = append(groups, safeReconciliationGroup{members: memberPositions, canonical: canonical})
|
||||
output.SourceRefs = order.Canonicalize(output.SourceRefs)
|
||||
output.ID = identity.DeriveID(output.Name)
|
||||
return output, nil
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
return groups
|
||||
|
||||
applied := application.Records()
|
||||
output := make([]normalizedRecord, len(applied))
|
||||
for index, record := range applied {
|
||||
output[index] = normalizedRecord{
|
||||
npc: record.Value(),
|
||||
inputIndexes: record.OriginalInputIndexes(),
|
||||
earliest: record.EarliestInputPosition(),
|
||||
}
|
||||
}
|
||||
warnings := make([]contracts.Warning, 0, len(application.AppliedGroups()))
|
||||
for _, event := range application.AppliedGroups() {
|
||||
provenance := event.Provenance()
|
||||
warnings = append(warnings, semanticDuplicateWarning(provenance, records[provenance.CanonicalPosition()]))
|
||||
}
|
||||
return output, warnings, nil
|
||||
}
|
||||
|
||||
func reconciliationIssues(assessment entityreconcile.Assessment) []string {
|
||||
issues := assessment.Issues()
|
||||
details := make([]string, len(issues))
|
||||
for index, issue := range issues {
|
||||
details[index] = fmt.Sprintf("group %d: %s", issue.GroupIndex, issue.Category)
|
||||
}
|
||||
return details
|
||||
}
|
||||
|
||||
func applySafeGroups(records []normalizedRecord, groups []safeReconciliationGroup, order shared.SourceRefOrder) ([]normalizedRecord, []contracts.Warning) {
|
||||
byMember := make(map[int]safeReconciliationGroup, len(groups)*2)
|
||||
for _, group := range groups {
|
||||
for _, member := range group.members {
|
||||
byMember[member] = group
|
||||
}
|
||||
}
|
||||
output := make([]normalizedRecord, 0, len(records)-len(groups))
|
||||
warnings := make([]contracts.Warning, 0, len(groups))
|
||||
for index, record := range records {
|
||||
group, grouped := byMember[index]
|
||||
if !grouped {
|
||||
output = append(output, cloneRecord(record))
|
||||
continue
|
||||
}
|
||||
if group.members[0] != index {
|
||||
continue
|
||||
}
|
||||
consolidated := consolidateSemanticGroup(records, group, order)
|
||||
output = append(output, consolidated)
|
||||
warnings = append(warnings, semanticDuplicateWarning(consolidated, records[group.canonical]))
|
||||
}
|
||||
return output, warnings
|
||||
}
|
||||
|
||||
func consolidateSemanticGroup(records []normalizedRecord, group safeReconciliationGroup, order shared.SourceRefOrder) normalizedRecord {
|
||||
output := cloneRecord(records[group.members[0]])
|
||||
output.npc.Name = records[group.canonical].npc.Name
|
||||
for _, member := range group.members[1:] {
|
||||
output.npc.SourceRefs = append(output.npc.SourceRefs, records[member].npc.SourceRefs...)
|
||||
output.inputIndexes = append(output.inputIndexes, records[member].inputIndexes...)
|
||||
if records[member].earliest < output.earliest {
|
||||
output.earliest = records[member].earliest
|
||||
}
|
||||
}
|
||||
output.inputIndexes = sortedUniqueIndexes(output.inputIndexes)
|
||||
output.npc.SourceRefs = order.Canonicalize(output.npc.SourceRefs)
|
||||
output.npc.ID = identity.DeriveID(output.npc.Name)
|
||||
return output
|
||||
}
|
||||
|
||||
func semanticDuplicateWarning(record normalizedRecord, canonical normalizedRecord) contracts.Warning {
|
||||
details := make([]string, 0, len(record.inputIndexes)+1)
|
||||
for _, inputIndex := range record.inputIndexes {
|
||||
func semanticDuplicateWarning(provenance semanticreconcile.GroupProvenance, canonical normalizedRecord) contracts.Warning {
|
||||
inputIndexes := provenance.OriginalInputIndexes()
|
||||
details := make([]string, 0, len(inputIndexes)+1)
|
||||
for _, inputIndex := range inputIndexes {
|
||||
details = append(details, fmt.Sprintf("input index %d", inputIndex))
|
||||
}
|
||||
if canonical.earliest != record.earliest {
|
||||
if canonical.earliest != provenance.EarliestInputPosition() {
|
||||
details = append(details, fmt.Sprintf("canonical display name from input index %d", canonical.earliest))
|
||||
}
|
||||
return contracts.Warning{
|
||||
Scope: npcScope(record.earliest),
|
||||
Scope: npcScope(provenance.EarliestInputPosition()),
|
||||
ReasonCode: ReasonCodeDuplicateNPCCollapsed,
|
||||
Message: diagnostics.Aggregate("semantic duplicate consolidation", details),
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"math"
|
||||
"reflect"
|
||||
"strings"
|
||||
@@ -11,9 +12,10 @@ import (
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/npcs/identity"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/diagnostics"
|
||||
)
|
||||
|
||||
func TestNormalizeSkipsSemanticCompletionWithoutTwoEligibleCandidates(t *testing.T) {
|
||||
@@ -31,7 +33,7 @@ func TestNormalizeSkipsSemanticCompletionWithoutTwoEligibleCandidates(t *testing
|
||||
}
|
||||
|
||||
func TestNormalizeAppliesSafeProposalAndUsesPrivateInputs(t *testing.T) {
|
||||
client := &recordingNPCNormalizerClient{response: `{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000002"}]}`}
|
||||
client := &recordingNPCNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2}]}`}
|
||||
normalizer := newNormalizer(t, client)
|
||||
doc := semanticDocument()
|
||||
input := dnd.NPCRegistry{NPCs: []dnd.NPC{
|
||||
@@ -64,17 +66,28 @@ func TestNormalizeAppliesSafeProposalAndUsesPrivateInputs(t *testing.T) {
|
||||
t.Fatalf("completion calls = %d, want one", len(client.requests))
|
||||
}
|
||||
completion := client.requests[0]
|
||||
if completion.StageName != Key || completion.PromptID != PromptID || completion.PromptVersion != entityreconcile.SchemaVersion || completion.ProfileID != request.LLMProfile || completion.SessionID != request.SessionID || len(completion.Inputs) != 2 {
|
||||
if completion.StageName != Key || completion.PromptID != PromptID || completion.PromptVersion != PromptVersion || completion.ProfileID != request.LLMProfile || completion.SessionID != request.SessionID || len(completion.Inputs) != 2 {
|
||||
t.Fatalf("completion request = %#v, want normalize request identity and exactly two inputs", completion)
|
||||
}
|
||||
encoded := string(completion.Inputs["candidates"].Content) + string(completion.Inputs["transcript"].Content)
|
||||
if strings.Contains(encoded, "npc:sha256:") || strings.Contains(encoded, doc.ID) {
|
||||
t.Fatalf("completion inputs leaked private identifiers: %s", encoded)
|
||||
}
|
||||
var visible struct {
|
||||
Candidates []struct {
|
||||
CandidateID int `json:"candidate_id"`
|
||||
} `json:"candidates"`
|
||||
}
|
||||
if err := json.Unmarshal(completion.Inputs["candidates"].Content, &visible); err != nil {
|
||||
t.Fatalf("decode candidates: %v", err)
|
||||
}
|
||||
if got := []int{visible.Candidates[0].CandidateID, visible.Candidates[1].CandidateID, visible.Candidates[2].CandidateID}; !reflect.DeepEqual(got, []int{1, 2, 3}) {
|
||||
t.Fatalf("candidate IDs = %v, want contiguous request-local handles", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeUnsafeProposalReturnsSafeRetryFallback(t *testing.T) {
|
||||
client := &recordingNPCNormalizerClient{response: `{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000002"},{"members":["candidate-000003","unknown"],"canonical":"candidate-000003"}]}`}
|
||||
client := &recordingNPCNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2},{"candidate_ids":[3,99],"canonical_candidate_id":3}]}`}
|
||||
normalizer := newNormalizer(t, client)
|
||||
doc := semanticDocument()
|
||||
input := dnd.NPCRegistry{NPCs: []dnd.NPC{
|
||||
@@ -137,7 +150,7 @@ func TestNormalizeRedactsContextMaterialFailures(t *testing.T) {
|
||||
request := normalizeRequestWithSource(input, doc)
|
||||
request.SourceInput = contracts.NewLLMInputMaterial("source", "application/json", []byte(transcript), "sha256:test", originPath)
|
||||
_, err := normalizer.Normalize(context.Background(), request)
|
||||
if err == nil || !strings.Contains(err.Error(), "build entity reconciliation context: invalid source metadata") {
|
||||
if err == nil || !strings.Contains(err.Error(), "build transcript material: invalid source metadata") {
|
||||
t.Fatalf("Normalize() error = %v; want content-safe context-material failure", err)
|
||||
}
|
||||
for _, forbidden := range []string{metadataKey, metadataValue, transcript, firstName, secondName, sourceID, originPath, "float64", "non-finite"} {
|
||||
@@ -152,8 +165,8 @@ func TestNormalizeRedactsContextMaterialFailures(t *testing.T) {
|
||||
|
||||
func TestNormalizeDoesNotAccumulateSafeGroupsAcrossAttempts(t *testing.T) {
|
||||
client := &recordingNPCNormalizerClient{responses: []string{
|
||||
`{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000002"},{"members":["candidate-000003","unknown"],"canonical":"candidate-000003"}]}`,
|
||||
`{"duplicate_groups":[{"members":["candidate-000001","candidate-000003"],"canonical":"candidate-000003"}]}`,
|
||||
`{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2},{"candidate_ids":[3,99],"canonical_candidate_id":3}]}`,
|
||||
`{"duplicate_groups":[{"candidate_ids":[1,3],"canonical_candidate_id":3}]}`,
|
||||
}}
|
||||
normalizer := newNormalizer(t, client)
|
||||
doc := semanticDocument()
|
||||
@@ -175,8 +188,8 @@ func TestNormalizeDoesNotAccumulateSafeGroupsAcrossAttempts(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeRetainsRecordsWhenAProposalUsesAnUnknownDescriptor(t *testing.T) {
|
||||
client := &recordingNPCNormalizerClient{response: `{"duplicate_groups":[{"members":["candidate-000001","unknown"],"canonical":"candidate-000001"}]}`}
|
||||
func TestNormalizeRetainsRecordsWhenAProposalUsesAnUnknownHandle(t *testing.T) {
|
||||
client := &recordingNPCNormalizerClient{response: `{"duplicate_groups":[{"candidate_ids":[1,99],"canonical_candidate_id":1}]}`}
|
||||
normalizer := newNormalizer(t, client)
|
||||
doc := semanticDocument()
|
||||
input := dnd.NPCRegistry{NPCs: []dnd.NPC{
|
||||
@@ -193,32 +206,52 @@ func TestNormalizeRetainsRecordsWhenAProposalUsesAnUnknownDescriptor(t *testing.
|
||||
}
|
||||
}
|
||||
|
||||
func TestReconciliationCandidatesKeepEqualDisplayNamesDistinct(t *testing.T) {
|
||||
func TestReconciliationCandidatesKeepEqualContextualDescriptorsDistinct(t *testing.T) {
|
||||
doc := semanticDocument()
|
||||
records := []normalizedRecord{
|
||||
{npc: dnd.NPC{Name: "The Guard", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 10, EndUnitID: 10}}}},
|
||||
{npc: dnd.NPC{Name: "The Guard", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 20}}}},
|
||||
}
|
||||
materials, ready, err := entityreconcile.BuildContext(doc, reconciliationCandidates(records), semanticContextRadius)
|
||||
if err != nil || !ready {
|
||||
t.Fatalf("BuildContext() = %#v, %t, %v; want ready keyed candidates", materials, ready, err)
|
||||
candidates, _, err := reconciliationInputs(records)
|
||||
if err != nil {
|
||||
t.Fatalf("reconciliationInputs() error = %v", err)
|
||||
}
|
||||
keys := materials.CandidateKeys()
|
||||
if !reflect.DeepEqual(keys, []string{"candidate-000001", "candidate-000002"}) || strings.Count(string(materials.Candidates.Content), `"The Guard"`) != 2 {
|
||||
t.Fatalf("candidate keys and inputs = %#v, %s; want distinct equal-display candidates", keys, materials.Candidates.Content)
|
||||
preparation, err := semanticreconcile.Prepare(doc, candidates, semanticreconcile.DefaultLimits())
|
||||
if err != nil || preparation.Disposition() != semanticreconcile.Ready {
|
||||
t.Fatalf("Prepare() = %#v, %v; want ready candidates", preparation, err)
|
||||
}
|
||||
var candidateInput struct {
|
||||
Candidates []entityreconcile.Selector `json:"candidates"`
|
||||
Candidates []struct {
|
||||
CandidateID int `json:"candidate_id"`
|
||||
Label string `json:"label"`
|
||||
} `json:"candidates"`
|
||||
}
|
||||
if err := json.Unmarshal(materials.Candidates.Content, &candidateInput); err != nil {
|
||||
if err := json.Unmarshal(preparation.Materials()["candidates"].Content, &candidateInput); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
assessment := materials.Assess(entityreconcile.ProposalResponse{DuplicateGroups: []entityreconcile.DuplicateGroup{{
|
||||
Members: candidateInput.Candidates, Canonical: candidateInput.Candidates[1],
|
||||
}}})
|
||||
groups := reconciliationGroups(assessment, keys)
|
||||
if assessment.DiscardedGroups() != 0 || len(groups) != 1 || groups[0].canonical != 1 {
|
||||
t.Fatalf("reconciliation = %#v, %#v; want distinct keyed group", assessment, groups)
|
||||
if len(candidateInput.Candidates) != 2 || candidateInput.Candidates[0].CandidateID != 1 || candidateInput.Candidates[1].CandidateID != 2 || candidateInput.Candidates[0].Label != "The Guard" || candidateInput.Candidates[1].Label != "The Guard" {
|
||||
t.Fatalf("candidate input = %#v, want distinct integer handles for equal descriptors", candidateInput)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNormalizeLimitSkipDoesNotCallLLMAndAddsBoundedFallbackWarning(t *testing.T) {
|
||||
client := &recordingNPCNormalizerClient{}
|
||||
normalizer := newNormalizer(t, client)
|
||||
doc := &source.SourceDocument{ID: "session", Units: []source.SourceUnit{{ID: 1, Kind: "speech", Text: "A crowded hall"}}}
|
||||
limit := semanticreconcile.DefaultLimits().MaximumCandidates
|
||||
input := dnd.NPCRegistry{NPCs: make([]dnd.NPC, limit+1)}
|
||||
for index := range input.NPCs {
|
||||
input.NPCs[index] = dnd.NPC{Name: fmt.Sprintf("Person %d", index), SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 1, EndUnitID: 1}}}
|
||||
}
|
||||
result, err := normalizer.Normalize(context.Background(), normalizeRequestWithSource(input, doc))
|
||||
if err != nil || result.Retry != nil {
|
||||
t.Fatalf("Normalize() = %#v, %v; want deterministic limit fallback", result, err)
|
||||
}
|
||||
if len(client.requests) != 0 || len(result.Value.NPCs) != limit+1 {
|
||||
t.Fatalf("completion calls = %d, NPCs = %d; want no call and all records", len(client.requests), len(result.Value.NPCs))
|
||||
}
|
||||
if !hasWarning(result.Warnings, ReasonCodeNPCSemanticReconciliationExhausted, "npcs") || len(result.Warnings) > diagnostics.MaxWarnings {
|
||||
t.Fatalf("warnings = %#v, want bounded reconciliation fallback", result.Warnings)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -35,7 +35,6 @@ import (
|
||||
npcnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/npcregistry"
|
||||
scenedescriptionnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/scenedescriptions"
|
||||
spellnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/spells"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/generic/merge/appendorder"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/generic/normalize/noop"
|
||||
)
|
||||
@@ -147,7 +146,6 @@ func registerModules(registries pipeline.Registries) error {
|
||||
|
||||
func registerPromptAssets(assets *llm.AssetRegistry) error {
|
||||
return runRegistrations([]registration{
|
||||
{name: "entity reconciliation schema assets", register: func() error { return entityreconcile.RegisterSchemaAssets(assets) }},
|
||||
{name: "scenes prompt assets", register: func() error { return scenes.RegisterPromptAssets(assets) }},
|
||||
{name: "spells prompt assets", register: func() error { return spellextract.RegisterPromptAssets(assets) }},
|
||||
{name: "npc registry prompt assets", register: func() error { return npcextract.RegisterPromptAssets(assets) }},
|
||||
|
||||
@@ -77,13 +77,6 @@ func TestRegisterAddsDNDFamily(t *testing.T) {
|
||||
if _, err := fs.ReadFile(fallbackFS, "dnd-extraction.yaml"); err != nil {
|
||||
t.Fatalf("fallback profile asset = %v, want registered D&D profile", err)
|
||||
}
|
||||
schemaFS, err := assets.SchemaFS()
|
||||
if err != nil {
|
||||
t.Fatalf("SchemaFS() error = %v", err)
|
||||
}
|
||||
if _, err := fs.ReadFile(schemaFS, "dnd_entity_reconcile_llm.v1.json"); err != nil {
|
||||
t.Fatalf("entity reconciliation schema asset = %v, want registered shared schema", err)
|
||||
}
|
||||
assertContainsKeys(t, "chunkers", registries.Chunkers.RegisteredKeys(), []string{"dnd/scenes"})
|
||||
assertContainsKeys(t, "extractors", registries.Extractors.RegisteredKeys(), []string{"dnd/spells", npcextract.Key, combatextract.Key, enemyeventextract.Key, itemoccurrenceextract.Key, itemregistryextract.Key, occurrenceextract.Key, scenedescriptionextract.Key, locationextract.Key, locationoccurrenceextract.Key})
|
||||
assertContainsKeys(t, "normalizers", registries.Normalizers.RegisteredKeys(), []string{spellnormalize.Key, npcnormalize.Key, combatnormalize.Key, enemyeventnormalize.Key, itemoccurrencenormalize.Key, itemregistrynormalize.Key, occurrencenormalize.Key, scenedescriptionnormalize.Key, locationnormalize.Key, locationoccurrencenormalize.Key, pipeline.DefaultNormalizeModule})
|
||||
|
||||
@@ -11,24 +11,24 @@ import (
|
||||
)
|
||||
|
||||
// PromptAssetManifest is the ordered set of assets that make up one prompt.
|
||||
// Module files are addressed in the owning module filesystem; shared files use
|
||||
// the names in sharedPromptPaths and are mounted beneath sharedassets.
|
||||
// Module files are addressed in the owning module filesystem. SharedFiles use
|
||||
// names in sharedPromptPaths, while ExternalSharedFiles are explicit
|
||||
// caller-owned descriptors. Both kinds are mounted beneath sharedassets.
|
||||
type PromptAssetManifest struct {
|
||||
ModuleDir string
|
||||
ModuleFiles []promptfs.ModulePromptFile
|
||||
SharedFiles []string
|
||||
ModuleDir string
|
||||
ModuleFiles []promptfs.ModulePromptFile
|
||||
SharedFiles []string
|
||||
ExternalSharedFiles []promptfs.SharedPromptFile
|
||||
}
|
||||
|
||||
var sharedPromptPaths = map[string]string{
|
||||
"common-dnd-system.md": "prompts/common-dnd-system.md",
|
||||
"common-dnd-extraction-evidence.md": "prompts/common-dnd-extraction-evidence.md",
|
||||
"common-dnd-identity.md": "prompts/common-dnd-identity.md",
|
||||
"common-dnd-transcript-full.md": "prompts/common-dnd-transcript-full.md",
|
||||
"common-dnd-transcript-chunk.md": "prompts/common-dnd-transcript-chunk.md",
|
||||
"common-dnd-transcript-windows.md": "prompts/common-dnd-transcript-windows.md",
|
||||
"common-dnd-references.md": "prompts/common-dnd-references.md",
|
||||
"common-dnd-npc-registry.md": "prompts/common-dnd-npc-registry.md",
|
||||
"common-dnd-entity-reconciliation.md": "prompts/common-dnd-entity-reconciliation.md",
|
||||
"common-dnd-system.md": "prompts/common-dnd-system.md",
|
||||
"common-dnd-extraction-evidence.md": "prompts/common-dnd-extraction-evidence.md",
|
||||
"common-dnd-identity.md": "prompts/common-dnd-identity.md",
|
||||
"common-dnd-transcript-full.md": "prompts/common-dnd-transcript-full.md",
|
||||
"common-dnd-transcript-chunk.md": "prompts/common-dnd-transcript-chunk.md",
|
||||
"common-dnd-references.md": "prompts/common-dnd-references.md",
|
||||
"common-dnd-npc-registry.md": "prompts/common-dnd-npc-registry.md",
|
||||
}
|
||||
|
||||
func sharedAssetFS() (fs.FS, error) {
|
||||
@@ -40,7 +40,7 @@ func sharedAssetFS() (fs.FS, error) {
|
||||
}
|
||||
|
||||
func (manifest PromptAssetManifest) PromptFS(moduleFS fs.FS) (fs.FS, error) {
|
||||
sharedFiles, err := resolveSharedPromptFiles(manifest.SharedFiles)
|
||||
sharedFiles, err := manifest.sharedPromptFiles()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
@@ -49,10 +49,14 @@ func (manifest PromptAssetManifest) PromptFS(moduleFS fs.FS) (fs.FS, error) {
|
||||
}
|
||||
|
||||
func (manifest PromptAssetManifest) Hash(moduleFS fs.FS) (string, error) {
|
||||
sharedFiles, err := resolveSharedPromptFiles(manifest.SharedFiles)
|
||||
sharedFiles, err := manifest.sharedPromptFiles()
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
moduleFiles := append([]promptfs.ModulePromptFile(nil), manifest.ModuleFiles...)
|
||||
if _, err := promptfs.ModulePromptFS(manifest.ModuleDir, moduleFS, moduleFiles, sharedFiles...); err != nil {
|
||||
return "", err
|
||||
}
|
||||
parts := make([]llm.AssetHashPart, 0, len(manifest.ModuleFiles)+len(sharedFiles))
|
||||
for _, file := range manifest.ModuleFiles {
|
||||
parts = append(parts, llm.AssetHashPart{FS: moduleFS, Path: file.Path})
|
||||
@@ -63,6 +67,14 @@ func (manifest PromptAssetManifest) Hash(moduleFS fs.FS) (string, error) {
|
||||
return llm.HashAssets(parts)
|
||||
}
|
||||
|
||||
func (manifest PromptAssetManifest) sharedPromptFiles() ([]promptfs.SharedPromptFile, error) {
|
||||
files, err := resolveSharedPromptFiles(manifest.SharedFiles)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return append(files, manifest.ExternalSharedFiles...), nil
|
||||
}
|
||||
|
||||
func resolveSharedPromptFiles(names []string) ([]promptfs.SharedPromptFile, error) {
|
||||
assets, err := sharedAssetFS()
|
||||
if err != nil {
|
||||
|
||||
@@ -22,7 +22,6 @@ func TestPromptAssetManifestPromptFS(t *testing.T) {
|
||||
"common-dnd-system.md",
|
||||
"common-dnd-transcript-full.md",
|
||||
"common-dnd-transcript-chunk.md",
|
||||
"common-dnd-transcript-windows.md",
|
||||
},
|
||||
}
|
||||
|
||||
@@ -51,7 +50,6 @@ func TestPromptAssetManifestPromptFS(t *testing.T) {
|
||||
"assets/prompts/dnd.test/sharedassets/common-dnd-system.md",
|
||||
"assets/prompts/dnd.test/sharedassets/common-dnd-transcript-full.md",
|
||||
"assets/prompts/dnd.test/sharedassets/common-dnd-transcript-chunk.md",
|
||||
"assets/prompts/dnd.test/sharedassets/common-dnd-transcript-windows.md",
|
||||
} {
|
||||
content, err := fs.ReadFile(fsys, path)
|
||||
if err != nil {
|
||||
@@ -207,3 +205,106 @@ func TestSharedPromptDescriptorsReturnFreshCopies(t *testing.T) {
|
||||
t.Fatalf("resolveSharedPromptFiles() reused descriptor state: first=%#v second=%#v", first, second)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPromptAssetManifestMountsExternalSharedFiles(t *testing.T) {
|
||||
external := fstest.MapFS{"core/protocol.md": {Data: []byte("integer protocol")}}
|
||||
manifest := PromptAssetManifest{
|
||||
ModuleDir: "dnd.test",
|
||||
ModuleFiles: []promptfs.ModulePromptFile{
|
||||
{Name: "prompt.yaml", Path: "prompts/prompt.yaml"},
|
||||
},
|
||||
ExternalSharedFiles: []promptfs.SharedPromptFile{
|
||||
{Name: "protocol.md", FS: external, Path: "core/protocol.md"},
|
||||
},
|
||||
}
|
||||
fys, err := manifest.PromptFS(fstest.MapFS{
|
||||
"prompts/prompt.yaml": {Data: []byte("id: dnd.test")},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("PromptFS() error = %v, want nil", err)
|
||||
}
|
||||
content, err := fs.ReadFile(fys, "assets/prompts/dnd.test/sharedassets/protocol.md")
|
||||
if err != nil || string(content) != "integer protocol" {
|
||||
t.Fatalf("mounted external protocol = %q, %v", content, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPromptAssetManifestHashIncludesExternalSharedFiles(t *testing.T) {
|
||||
moduleFS := fstest.MapFS{"prompts/prompt.yaml": {Data: []byte("id: dnd.test")}}
|
||||
external := fstest.MapFS{"core/protocol.md": {Data: []byte("integer protocol")}}
|
||||
manifest := PromptAssetManifest{
|
||||
ModuleDir: "dnd.test",
|
||||
ModuleFiles: []promptfs.ModulePromptFile{
|
||||
{Name: "prompt.yaml", Path: "prompts/prompt.yaml"},
|
||||
},
|
||||
ExternalSharedFiles: []promptfs.SharedPromptFile{
|
||||
{Name: "protocol.md", FS: external, Path: "core/protocol.md"},
|
||||
},
|
||||
}
|
||||
got, err := manifest.Hash(moduleFS)
|
||||
if err != nil {
|
||||
t.Fatalf("Hash() error = %v, want nil", err)
|
||||
}
|
||||
want, err := llm.HashAssets([]llm.AssetHashPart{
|
||||
{FS: moduleFS, Path: "prompts/prompt.yaml"},
|
||||
{FS: external, Path: "core/protocol.md"},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got != want {
|
||||
t.Fatalf("Hash() = %q, want external-aware hash %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPromptAssetManifestRejectsInvalidExternalSharedFiles(t *testing.T) {
|
||||
moduleFS := fstest.MapFS{"prompts/prompt.yaml": {Data: []byte("id: dnd.test")}}
|
||||
tests := []struct {
|
||||
name string
|
||||
external []promptfs.SharedPromptFile
|
||||
shared []string
|
||||
want string
|
||||
}{
|
||||
{
|
||||
name: "missing",
|
||||
external: []promptfs.SharedPromptFile{
|
||||
{Name: "protocol.md", FS: fstest.MapFS{}, Path: "core/protocol.md"},
|
||||
},
|
||||
want: "read shared prompt asset core/protocol.md",
|
||||
},
|
||||
{
|
||||
name: "duplicate external destinations",
|
||||
external: []promptfs.SharedPromptFile{
|
||||
{Name: "protocol.md", FS: fstest.MapFS{"a.md": {Data: []byte("a")}}, Path: "a.md"},
|
||||
{Name: " protocol.md ", FS: fstest.MapFS{"b.md": {Data: []byte("b")}}, Path: "b.md"},
|
||||
},
|
||||
want: `duplicate sharedassets prompt destination "assets/prompts/dnd.test/sharedassets/protocol.md"`,
|
||||
},
|
||||
{
|
||||
name: "duplicate named and external destinations",
|
||||
shared: []string{"common-dnd-system.md"},
|
||||
external: []promptfs.SharedPromptFile{
|
||||
{Name: "common-dnd-system.md", FS: fstest.MapFS{"system.md": {Data: []byte("external")}}, Path: "system.md"},
|
||||
},
|
||||
want: `duplicate sharedassets prompt destination "assets/prompts/dnd.test/sharedassets/common-dnd-system.md"`,
|
||||
},
|
||||
}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
manifest := PromptAssetManifest{
|
||||
ModuleDir: "dnd.test",
|
||||
ModuleFiles: []promptfs.ModulePromptFile{
|
||||
{Name: "prompt.yaml", Path: "prompts/prompt.yaml"},
|
||||
},
|
||||
SharedFiles: test.shared,
|
||||
ExternalSharedFiles: test.external,
|
||||
}
|
||||
if _, err := manifest.PromptFS(moduleFS); err == nil || !strings.Contains(err.Error(), test.want) {
|
||||
t.Fatalf("PromptFS() error = %v, want %q", err, test.want)
|
||||
}
|
||||
if _, err := manifest.Hash(moduleFS); err == nil || !strings.Contains(err.Error(), test.want) {
|
||||
t.Fatalf("Hash() error = %v, want %q", err, test.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,292 +0,0 @@
|
||||
// Package entityreconcile provides safe, D&D-specific duplicate proposal
|
||||
// materials shared by entity normalizers.
|
||||
package entityreconcile
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"sort"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
)
|
||||
|
||||
const candidateKeyFormat = "candidate-%06d"
|
||||
|
||||
// Candidate is one domain-neutral entity candidate supplied by a normalizer.
|
||||
// BuildContext never retains or mutates its source references.
|
||||
type Candidate struct {
|
||||
Name string
|
||||
SourceRefs []source.SourceRef
|
||||
}
|
||||
|
||||
// Selector identifies one candidate through its canonical name and source-free
|
||||
// evidence ranges. It is the complete model-facing candidate descriptor.
|
||||
type Selector struct {
|
||||
Name string `json:"name"`
|
||||
SourceRefs []SourceRange `json:"source_refs"`
|
||||
}
|
||||
|
||||
// Clone returns an owned copy of the selector.
|
||||
func (s Selector) Clone() Selector {
|
||||
s.SourceRefs = cloneSourceRanges(s.SourceRefs)
|
||||
return s
|
||||
}
|
||||
|
||||
// SourceRange is a source-free evidence coordinate used in a selector.
|
||||
type SourceRange struct {
|
||||
StartUnitID int `json:"start_unit_id"`
|
||||
EndUnitID int `json:"end_unit_id"`
|
||||
}
|
||||
|
||||
// Materials contains owned prompt inputs and opaque candidate-key mappings.
|
||||
type Materials struct {
|
||||
Candidates contracts.LLMInputMaterial
|
||||
Transcript contracts.LLMInputMaterial
|
||||
|
||||
candidateKeys []string
|
||||
eligible map[string]struct{}
|
||||
keyBySelector map[string]string
|
||||
collidedSelectors map[string]struct{}
|
||||
}
|
||||
|
||||
// CandidateKeys returns all deterministic keys in candidate input order.
|
||||
func (m Materials) CandidateKeys() []string {
|
||||
return append([]string(nil), m.candidateKeys...)
|
||||
}
|
||||
|
||||
// EligibleCandidateKeys returns only candidates whose evidence safely produced
|
||||
// transcript context, preserving candidate input order.
|
||||
func (m Materials) EligibleCandidateKeys() []string {
|
||||
keys := make([]string, 0, len(m.eligible))
|
||||
for _, key := range m.candidateKeys {
|
||||
if _, ok := m.eligible[key]; ok {
|
||||
keys = append(keys, key)
|
||||
}
|
||||
}
|
||||
return keys
|
||||
}
|
||||
|
||||
type candidateInput struct {
|
||||
Candidates []Selector `json:"candidates"`
|
||||
}
|
||||
|
||||
type transcriptInput struct {
|
||||
Windows []transcriptWindow `json:"windows"`
|
||||
}
|
||||
|
||||
type transcriptWindow struct {
|
||||
Units []transcriptUnit `json:"units"`
|
||||
}
|
||||
|
||||
type transcriptUnit struct {
|
||||
ID int `json:"id"`
|
||||
Kind string `json:"kind"`
|
||||
Text string `json:"text"`
|
||||
Metadata map[string]any `json:"metadata,omitempty"`
|
||||
Cited bool `json:"cited"`
|
||||
}
|
||||
|
||||
type sourceInterval struct {
|
||||
start int
|
||||
end int
|
||||
}
|
||||
|
||||
// BuildContext constructs bounded, source-ordered prompt inputs. It returns
|
||||
// ready=false when fewer than two candidates have safe context.
|
||||
func BuildContext(doc *source.SourceDocument, candidates []Candidate, radius int) (Materials, bool, error) {
|
||||
if radius < 0 {
|
||||
return Materials{}, false, fmt.Errorf("build entity reconciliation context: radius must not be negative")
|
||||
}
|
||||
materials := Materials{
|
||||
candidateKeys: make([]string, len(candidates)),
|
||||
eligible: make(map[string]struct{}),
|
||||
keyBySelector: make(map[string]string),
|
||||
collidedSelectors: make(map[string]struct{}),
|
||||
}
|
||||
for index := range candidates {
|
||||
key := fmt.Sprintf(candidateKeyFormat, index+1)
|
||||
materials.candidateKeys[index] = key
|
||||
}
|
||||
if doc == nil {
|
||||
return materials, false, nil
|
||||
}
|
||||
|
||||
index := source.NewDocumentIndex(doc)
|
||||
type preparedCandidate struct {
|
||||
key string
|
||||
selector Selector
|
||||
intervals []sourceInterval
|
||||
lookupKey string
|
||||
}
|
||||
prepared := make([]preparedCandidate, 0, len(candidates))
|
||||
selectorCounts := make(map[string]int, len(candidates))
|
||||
views := make([]Selector, 0, len(candidates))
|
||||
intervals := make([]sourceInterval, 0)
|
||||
cited := make([]bool, len(doc.Units))
|
||||
for candidateIndex, candidate := range candidates {
|
||||
references, candidateIntervals, valid := candidateReferences(index, candidate.SourceRefs)
|
||||
if !valid {
|
||||
continue
|
||||
}
|
||||
key := materials.candidateKeys[candidateIndex]
|
||||
selector := Selector{Name: candidate.Name, SourceRefs: references}
|
||||
lookupKey, err := selectorLookupKey(selector)
|
||||
if err != nil {
|
||||
return Materials{}, false, fmt.Errorf("build entity reconciliation context: invalid candidate material")
|
||||
}
|
||||
prepared = append(prepared, preparedCandidate{key: key, selector: selector, intervals: candidateIntervals, lookupKey: lookupKey})
|
||||
selectorCounts[lookupKey]++
|
||||
}
|
||||
for _, candidate := range prepared {
|
||||
if selectorCounts[candidate.lookupKey] != 1 {
|
||||
materials.collidedSelectors[candidate.lookupKey] = struct{}{}
|
||||
continue
|
||||
}
|
||||
materials.eligible[candidate.key] = struct{}{}
|
||||
materials.keyBySelector[candidate.lookupKey] = candidate.key
|
||||
views = append(views, candidate.selector.Clone())
|
||||
for _, interval := range candidate.intervals {
|
||||
for position := interval.start; position <= interval.end; position++ {
|
||||
cited[position] = true
|
||||
}
|
||||
intervals = append(intervals, sourceInterval{
|
||||
start: maxInt(0, interval.start-radius),
|
||||
end: minInt(len(doc.Units)-1, interval.end+radius),
|
||||
})
|
||||
}
|
||||
}
|
||||
if len(views) < 2 {
|
||||
return materials, false, nil
|
||||
}
|
||||
|
||||
windows, err := contextWindows(doc.Units, coalesceIntervals(intervals), cited)
|
||||
if err != nil {
|
||||
return Materials{}, false, fmt.Errorf("build entity reconciliation context: invalid source metadata")
|
||||
}
|
||||
candidateContent, err := json.Marshal(candidateInput{Candidates: views})
|
||||
if err != nil {
|
||||
return Materials{}, false, fmt.Errorf("build entity reconciliation context: invalid candidate material")
|
||||
}
|
||||
transcriptContent, err := json.Marshal(transcriptInput{Windows: windows})
|
||||
if err != nil {
|
||||
return Materials{}, false, fmt.Errorf("build entity reconciliation context: invalid transcript material")
|
||||
}
|
||||
materials.Candidates = newInputMaterial("candidates", candidateContent)
|
||||
materials.Transcript = newInputMaterial("transcript", transcriptContent)
|
||||
return materials, true, nil
|
||||
}
|
||||
|
||||
func candidateReferences(index source.DocumentIndex, refs []source.SourceRef) ([]SourceRange, []sourceInterval, bool) {
|
||||
if len(refs) == 0 {
|
||||
return nil, nil, false
|
||||
}
|
||||
type referencedInterval struct {
|
||||
reference SourceRange
|
||||
interval sourceInterval
|
||||
}
|
||||
prepared := make([]referencedInterval, 0, len(refs))
|
||||
for _, ref := range refs {
|
||||
if err := index.ValidateRef(ref); err != nil {
|
||||
return nil, nil, false
|
||||
}
|
||||
start, _ := index.Position(ref.StartUnitID)
|
||||
end, _ := index.Position(ref.EndUnitID)
|
||||
prepared = append(prepared, referencedInterval{reference: SourceRange{StartUnitID: ref.StartUnitID, EndUnitID: ref.EndUnitID}, interval: sourceInterval{start: start, end: end}})
|
||||
}
|
||||
sort.Slice(prepared, func(left, right int) bool {
|
||||
if prepared[left].interval.start != prepared[right].interval.start {
|
||||
return prepared[left].interval.start < prepared[right].interval.start
|
||||
}
|
||||
return prepared[left].interval.end < prepared[right].interval.end
|
||||
})
|
||||
references := make([]SourceRange, 0, len(prepared))
|
||||
intervals := make([]sourceInterval, 0, len(prepared))
|
||||
for _, item := range prepared {
|
||||
if len(references) > 0 && references[len(references)-1] == item.reference {
|
||||
continue
|
||||
}
|
||||
references = append(references, item.reference)
|
||||
intervals = append(intervals, item.interval)
|
||||
}
|
||||
return references, intervals, true
|
||||
}
|
||||
|
||||
func selectorLookupKey(selector Selector) (string, error) {
|
||||
content, err := json.Marshal(selector.Clone())
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
return string(content), nil
|
||||
}
|
||||
|
||||
func cloneSourceRanges(values []SourceRange) []SourceRange {
|
||||
if len(values) == 0 {
|
||||
return []SourceRange{}
|
||||
}
|
||||
return append([]SourceRange(nil), values...)
|
||||
}
|
||||
|
||||
func coalesceIntervals(intervals []sourceInterval) []sourceInterval {
|
||||
if len(intervals) == 0 {
|
||||
return nil
|
||||
}
|
||||
ordered := append([]sourceInterval(nil), intervals...)
|
||||
sort.Slice(ordered, func(left, right int) bool {
|
||||
if ordered[left].start != ordered[right].start {
|
||||
return ordered[left].start < ordered[right].start
|
||||
}
|
||||
return ordered[left].end < ordered[right].end
|
||||
})
|
||||
coalesced := make([]sourceInterval, 0, len(ordered))
|
||||
for _, interval := range ordered {
|
||||
if len(coalesced) == 0 || interval.start > coalesced[len(coalesced)-1].end+1 {
|
||||
coalesced = append(coalesced, interval)
|
||||
continue
|
||||
}
|
||||
if interval.end > coalesced[len(coalesced)-1].end {
|
||||
coalesced[len(coalesced)-1].end = interval.end
|
||||
}
|
||||
}
|
||||
return coalesced
|
||||
}
|
||||
|
||||
func contextWindows(units []source.SourceUnit, intervals []sourceInterval, cited []bool) ([]transcriptWindow, error) {
|
||||
windows := make([]transcriptWindow, 0, len(intervals))
|
||||
for _, interval := range intervals {
|
||||
window := transcriptWindow{Units: make([]transcriptUnit, 0, interval.end-interval.start+1)}
|
||||
for position := interval.start; position <= interval.end; position++ {
|
||||
unit := units[position]
|
||||
metadata, err := source.CloneMetadata(unit.Metadata)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
window.Units = append(window.Units, transcriptUnit{
|
||||
ID: unit.ID, Kind: unit.Kind, Text: unit.Text, Metadata: metadata, Cited: cited[position],
|
||||
})
|
||||
}
|
||||
windows = append(windows, window)
|
||||
}
|
||||
return windows, nil
|
||||
}
|
||||
|
||||
func newInputMaterial(name string, content []byte) contracts.LLMInputMaterial {
|
||||
digest := sha256.Sum256(content)
|
||||
return contracts.NewLLMInputMaterial(name, "application/json", content, "sha256:"+hex.EncodeToString(digest[:]), "")
|
||||
}
|
||||
|
||||
func minInt(left, right int) int {
|
||||
if left < right {
|
||||
return left
|
||||
}
|
||||
return right
|
||||
}
|
||||
|
||||
func maxInt(left, right int) int {
|
||||
if left > right {
|
||||
return left
|
||||
}
|
||||
return right
|
||||
}
|
||||
@@ -1,328 +0,0 @@
|
||||
package entityreconcile
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"io/fs"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/core/source"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"github.com/santhosh-tekuri/jsonschema/v6"
|
||||
)
|
||||
|
||||
func TestBuildContextUsesContextualSelectorsSourceOrderAndOwnedData(t *testing.T) {
|
||||
doc := &source.SourceDocument{ID: "session", Units: []source.SourceUnit{
|
||||
{ID: 40, Kind: "narration", Text: "zero"},
|
||||
{ID: 10, Kind: "speech", Text: "one", Metadata: map[string]any{"speaker": map[string]any{"name": "Mira"}}},
|
||||
{ID: 70, Kind: "speech", Text: "two"},
|
||||
{ID: 20, Kind: "narration", Text: "three"},
|
||||
{ID: 90, Kind: "speech", Text: "four"},
|
||||
}}
|
||||
candidates := []Candidate{
|
||||
{Name: "The Tavern", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 10, EndUnitID: 20}}},
|
||||
{Name: "The Tavern", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 90, EndUnitID: 90}}},
|
||||
{Name: "Broken", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 10}}},
|
||||
}
|
||||
before := cloneCandidates(candidates)
|
||||
|
||||
materials, ready, err := BuildContext(doc, candidates, 1)
|
||||
if err != nil || !ready {
|
||||
t.Fatalf("BuildContext() = %#v, %t, %v; want ready materials", materials, ready, err)
|
||||
}
|
||||
if !reflect.DeepEqual(candidates, before) {
|
||||
t.Fatalf("BuildContext() mutated candidates: %#v", candidates)
|
||||
}
|
||||
if got, want := materials.CandidateKeys(), []string{"candidate-000001", "candidate-000002", "candidate-000003"}; !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("CandidateKeys() = %#v, want %#v", got, want)
|
||||
}
|
||||
if got, want := materials.EligibleCandidateKeys(), []string{"candidate-000001", "candidate-000002"}; !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("EligibleCandidateKeys() = %#v, want %#v", got, want)
|
||||
}
|
||||
if !json.Valid(materials.Candidates.Content) || !json.Valid(materials.Transcript.Content) {
|
||||
t.Fatalf("prompt materials are not JSON: %#v", materials)
|
||||
}
|
||||
if strings.Contains(string(materials.Candidates.Content), doc.ID) {
|
||||
t.Fatalf("candidate material leaked source identity: %s", materials.Candidates.Content)
|
||||
}
|
||||
|
||||
var candidatePayload candidateInput
|
||||
if err := json.Unmarshal(materials.Candidates.Content, &candidatePayload); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(candidatePayload.Candidates) != 2 || candidatePayload.Candidates[0].Name != candidatePayload.Candidates[1].Name || strings.Contains(string(materials.Candidates.Content), "candidate-") {
|
||||
t.Fatalf("candidate payload = %#v, want contextual descriptors without keys", candidatePayload)
|
||||
}
|
||||
if got := candidatePayload.Candidates[0].SourceRefs[0]; got != (SourceRange{StartUnitID: 10, EndUnitID: 20}) {
|
||||
t.Fatalf("candidate reference = %#v", got)
|
||||
}
|
||||
|
||||
var transcript transcriptInput
|
||||
if err := json.Unmarshal(materials.Transcript.Content, &transcript); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(transcript.Windows) != 1 || len(transcript.Windows[0].Units) != 5 {
|
||||
t.Fatalf("windows = %#v, want one bounded coalesced window", transcript.Windows)
|
||||
}
|
||||
units := transcript.Windows[0].Units
|
||||
for index, wantID := range []int{40, 10, 70, 20, 90} {
|
||||
if units[index].ID != wantID {
|
||||
t.Fatalf("window unit %d = %d, want source-order %d", index, units[index].ID, wantID)
|
||||
}
|
||||
}
|
||||
if units[0].Cited || !units[1].Cited || !units[2].Cited || !units[3].Cited || !units[4].Cited {
|
||||
t.Fatalf("citation flags = %#v", units)
|
||||
}
|
||||
|
||||
windows, err := contextWindows(doc.Units, []sourceInterval{{start: 1, end: 1}}, make([]bool, len(doc.Units)))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
windows[0].Units[0].Metadata["speaker"].(map[string]any)["name"] = "changed"
|
||||
if doc.Units[1].Metadata["speaker"].(map[string]any)["name"] != "Mira" {
|
||||
t.Fatal("context metadata aliases source document")
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildContextExcludesCollidingDescriptors(t *testing.T) {
|
||||
doc := &source.SourceDocument{ID: "session", Units: []source.SourceUnit{{ID: 1}, {ID: 2}, {ID: 3}}}
|
||||
candidates := []Candidate{
|
||||
{Name: "The Tavern", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 1, EndUnitID: 1}}},
|
||||
{Name: "The Tavern", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 1, EndUnitID: 1}}},
|
||||
{Name: "The Market", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 3, EndUnitID: 3}}},
|
||||
}
|
||||
materials, ready, err := BuildContext(doc, candidates, 0)
|
||||
if err != nil || ready || len(materials.EligibleCandidateKeys()) != 1 || strings.Contains(string(materials.Candidates.Content), "The Tavern") {
|
||||
t.Fatalf("BuildContext() = %#v, %t, %v", materials, ready, err)
|
||||
}
|
||||
assessment := materials.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{{
|
||||
Members: []Selector{{Name: "The Tavern", SourceRefs: []SourceRange{{StartUnitID: 1, EndUnitID: 1}}}, {Name: "The Market", SourceRefs: []SourceRange{{StartUnitID: 3, EndUnitID: 3}}}},
|
||||
Canonical: Selector{Name: "The Market", SourceRefs: []SourceRange{{StartUnitID: 3, EndUnitID: 3}}},
|
||||
}}})
|
||||
if !hasIssue(assessment.Issues(), "member_ineligible") {
|
||||
t.Fatalf("Assess() issues = %#v, want collided descriptor rejection", assessment.Issues())
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssessmentRejectsPartialAndReorderedDescriptors(t *testing.T) {
|
||||
doc := &source.SourceDocument{ID: "session", Units: []source.SourceUnit{{ID: 10}, {ID: 20}, {ID: 30}}}
|
||||
materials, ready, err := BuildContext(doc, []Candidate{
|
||||
{Name: "The Tavern", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 20, EndUnitID: 20}, {SourceID: doc.ID, StartUnitID: 10, EndUnitID: 10}}},
|
||||
{Name: "The Market", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 30, EndUnitID: 30}}},
|
||||
}, 0)
|
||||
if err != nil || !ready {
|
||||
t.Fatalf("BuildContext() = %#v, %t, %v", materials, ready, err)
|
||||
}
|
||||
selectors := materialSelectors(t, materials)
|
||||
for _, refs := range [][]SourceRange{
|
||||
{{StartUnitID: 10, EndUnitID: 10}},
|
||||
{{StartUnitID: 20, EndUnitID: 20}, {StartUnitID: 10, EndUnitID: 10}},
|
||||
} {
|
||||
assessment := materials.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{{
|
||||
Members: []Selector{{Name: "The Tavern", SourceRefs: refs}, selectors[1]},
|
||||
Canonical: selectors[1],
|
||||
}}})
|
||||
if !hasIssue(assessment.Issues(), "member_unknown") {
|
||||
t.Fatalf("Assess(%#v) issues = %#v, want descriptor mismatch rejection", refs, assessment.Issues())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildContextExcludesUnsafeReferencesAndCoalescesAdjacentWindows(t *testing.T) {
|
||||
doc := &source.SourceDocument{ID: "session", Units: []source.SourceUnit{{ID: 9}, {ID: 3}, {ID: 8}, {ID: 1}, {ID: 7}}}
|
||||
candidates := []Candidate{
|
||||
{Name: "One", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 3, EndUnitID: 3}}},
|
||||
{Name: "Two", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 8, EndUnitID: 8}}},
|
||||
{Name: "Missing", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: 99, EndUnitID: 99}}},
|
||||
{Name: "Foreign", SourceRefs: []source.SourceRef{{SourceID: "other", StartUnitID: 1, EndUnitID: 1}}},
|
||||
{Name: "Blank"},
|
||||
}
|
||||
materials, ready, err := BuildContext(doc, candidates, 0)
|
||||
if err != nil || !ready {
|
||||
t.Fatalf("BuildContext() error = %v, ready = %t", err, ready)
|
||||
}
|
||||
if got, want := materials.EligibleCandidateKeys(), []string{"candidate-000001", "candidate-000002"}; !reflect.DeepEqual(got, want) {
|
||||
t.Fatalf("EligibleCandidateKeys() = %#v, want %#v", got, want)
|
||||
}
|
||||
var transcript transcriptInput
|
||||
if err := json.Unmarshal(materials.Transcript.Content, &transcript); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(transcript.Windows) != 1 || len(transcript.Windows[0].Units) != 2 || transcript.Windows[0].Units[0].ID != 3 || transcript.Windows[0].Units[1].ID != 8 {
|
||||
t.Fatalf("windows = %#v, want adjacent document-order units coalesced", transcript.Windows)
|
||||
}
|
||||
if _, ready, err := BuildContext(doc, candidates, -1); err == nil || ready {
|
||||
t.Fatalf("BuildContext(radius=-1) = ready %t, err %v", ready, err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssessmentRejectsEveryUnsafeProposalCategory(t *testing.T) {
|
||||
materials := preparedMaterials(t, 4, true)
|
||||
selectors := materialSelectors(t, materials)
|
||||
unsafe := []struct {
|
||||
name string
|
||||
response ProposalResponse
|
||||
category string
|
||||
}{
|
||||
{"blank member", ProposalResponse{DuplicateGroups: []DuplicateGroup{{Members: []Selector{{}, selectors[1]}, Canonical: selectors[1]}}}, "member_blank"},
|
||||
{"unknown member", ProposalResponse{DuplicateGroups: []DuplicateGroup{{Members: []Selector{{Name: "unknown", SourceRefs: []SourceRange{{StartUnitID: 99, EndUnitID: 99}}}, selectors[1]}, Canonical: selectors[1]}}}, "member_unknown"},
|
||||
{"repeated member", ProposalResponse{DuplicateGroups: []DuplicateGroup{{Members: []Selector{selectors[0], selectors[0]}, Canonical: selectors[0]}}}, "repeated_member"},
|
||||
{"too small", ProposalResponse{DuplicateGroups: []DuplicateGroup{{Members: []Selector{selectors[0]}, Canonical: selectors[0]}}}, "fewer_than_two_members"},
|
||||
{"canonical blank", ProposalResponse{DuplicateGroups: []DuplicateGroup{{Members: []Selector{selectors[0], selectors[1]}, Canonical: Selector{}}}}, "canonical_blank"},
|
||||
{"canonical not member", ProposalResponse{DuplicateGroups: []DuplicateGroup{{Members: []Selector{selectors[0], selectors[1]}, Canonical: selectors[2]}}}, "canonical_not_member"},
|
||||
{"overlapping", ProposalResponse{DuplicateGroups: []DuplicateGroup{
|
||||
{Members: []Selector{selectors[0], selectors[1]}, Canonical: selectors[0]},
|
||||
{Members: []Selector{selectors[1], selectors[2]}, Canonical: selectors[2]},
|
||||
}}, "overlapping_member"},
|
||||
}
|
||||
for _, test := range unsafe {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
assessment := materials.Assess(test.response)
|
||||
if len(assessment.SafeGroups()) != 0 || assessment.DiscardedGroups() != len(test.response.DuplicateGroups) || !hasIssue(assessment.Issues(), test.category) {
|
||||
t.Fatalf("Assess() = groups %#v discarded %d issues %#v; want %q rejection", assessment.SafeGroups(), assessment.DiscardedGroups(), assessment.Issues(), test.category)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssessmentReturnsNonOverlappingSafeGroupsAndDefensiveCopies(t *testing.T) {
|
||||
materials := preparedMaterials(t, 4, false)
|
||||
keys := materials.CandidateKeys()
|
||||
selectors := materialSelectors(t, materials)
|
||||
assessment := materials.Assess(ProposalResponse{DuplicateGroups: []DuplicateGroup{
|
||||
{Members: []Selector{selectors[1], selectors[0]}, Canonical: selectors[1]},
|
||||
{Members: []Selector{selectors[3], selectors[2]}, Canonical: selectors[2]},
|
||||
}})
|
||||
groups := assessment.SafeGroups()
|
||||
if assessment.DiscardedGroups() != 0 || len(assessment.Issues()) != 0 || len(groups) != 2 {
|
||||
t.Fatalf("assessment = %#v, %d, %#v", groups, assessment.DiscardedGroups(), assessment.Issues())
|
||||
}
|
||||
if got, want := groups[0].Members(), []string{keys[0], keys[1]}; !reflect.DeepEqual(got, want) || groups[0].Canonical() != keys[1] {
|
||||
t.Fatalf("first safe group = %#v / %q", got, groups[0].Canonical())
|
||||
}
|
||||
keys[0] = "changed"
|
||||
if materials.CandidateKeys()[0] == "changed" {
|
||||
t.Fatal("CandidateKeys() exposed retained keys")
|
||||
}
|
||||
members := groups[0].Members()
|
||||
members[0] = "changed"
|
||||
if groups[0].Members()[0] == "changed" || assessment.SafeGroups()[0].Members()[0] == "changed" {
|
||||
t.Fatal("SafeGroups() exposed retained members")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSharedResponseSchemaIsPrivateStrictAndRegisterableOnce(t *testing.T) {
|
||||
schema, err := LoadResponseSchema()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if schema.Key != ResponseSchemaKey || schema.ID != ResponseSchemaID || schema.Name != ResponseSchemaName || schema.Version != SchemaVersion || !strings.HasPrefix(schema.SHA256, "sha256:") || !json.Valid(schema.JSONSchema) {
|
||||
t.Fatalf("schema = %#v", schema)
|
||||
}
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
value any
|
||||
valid bool
|
||||
}{
|
||||
{"empty groups", map[string]any{"duplicate_groups": []any{}}, true},
|
||||
{"semantic proposal problem", map[string]any{"duplicate_groups": []any{map[string]any{"members": []any{map[string]any{"name": "Mira", "source_refs": []any{map[string]any{"start_unit_id": 1, "end_unit_id": 1}}}}, "canonical": map[string]any{"name": "Mira", "source_refs": []any{map[string]any{"start_unit_id": 1, "end_unit_id": 1}}}}}}, true},
|
||||
{"missing groups", map[string]any{}, false},
|
||||
{"unknown top level", map[string]any{"duplicate_groups": []any{}, "extra": true}, false},
|
||||
{"replacement name", map[string]any{"duplicate_groups": []any{map[string]any{"members": []any{}, "canonical": map[string]any{"name": "Mira", "source_refs": []any{}}, "name": "replacement"}}}, false},
|
||||
{"missing selector evidence", map[string]any{"duplicate_groups": []any{map[string]any{"members": []any{}, "canonical": map[string]any{"name": "Mira"}}}}, false},
|
||||
{"invalid range", map[string]any{"duplicate_groups": []any{map[string]any{"members": []any{map[string]any{"name": "Mira", "source_refs": []any{map[string]any{"start_unit_id": 0, "end_unit_id": 1}}}}, "canonical": map[string]any{"name": "Mira", "source_refs": []any{}}}}}, false},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
content, err := json.Marshal(test.value)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
err = validateSchema(content, schema.JSONSchema)
|
||||
if (err == nil) != test.valid {
|
||||
t.Fatalf("validateSchema() error = %v, want valid=%t", err, test.valid)
|
||||
}
|
||||
})
|
||||
}
|
||||
first := schema.JSONSchema
|
||||
first[0] = '['
|
||||
second, err := LoadResponseSchema()
|
||||
if err != nil || !json.Valid(second.JSONSchema) || bytes.Equal(first, second.JSONSchema) {
|
||||
t.Fatalf("LoadResponseSchema() returned shared content: %s, %v", second.JSONSchema, err)
|
||||
}
|
||||
registry := llm.NewAssetRegistry()
|
||||
if err := RegisterSchemaAssets(registry); err != nil {
|
||||
t.Fatalf("RegisterSchemaAssets() error = %v", err)
|
||||
}
|
||||
if schemaFS, err := registry.SchemaFS(); err != nil {
|
||||
t.Fatalf("SchemaFS() error = %v", err)
|
||||
} else if content, err := fs.ReadFile(schemaFS, "dnd_entity_reconcile_llm.v1.json"); err != nil || !json.Valid(content) {
|
||||
t.Fatalf("shared schema asset = %s, %v", content, err)
|
||||
}
|
||||
}
|
||||
|
||||
func preparedMaterials(t *testing.T, count int, includeIneligible bool) Materials {
|
||||
t.Helper()
|
||||
doc := &source.SourceDocument{ID: "session", Units: make([]source.SourceUnit, count)}
|
||||
candidates := make([]Candidate, count)
|
||||
for index := range candidates {
|
||||
doc.Units[index] = source.SourceUnit{ID: index + 1, Text: "unit"}
|
||||
candidates[index] = Candidate{Name: "same display name", SourceRefs: []source.SourceRef{{SourceID: doc.ID, StartUnitID: index + 1, EndUnitID: index + 1}}}
|
||||
}
|
||||
if includeIneligible && count > 3 {
|
||||
candidates[3].SourceRefs = []source.SourceRef{{SourceID: "other", StartUnitID: 1, EndUnitID: 1}}
|
||||
}
|
||||
materials, ready, err := BuildContext(doc, candidates, 0)
|
||||
if err != nil || !ready {
|
||||
t.Fatalf("BuildContext() = %#v, %t, %v", materials, ready, err)
|
||||
}
|
||||
return materials
|
||||
}
|
||||
|
||||
func materialSelectors(t *testing.T, materials Materials) []Selector {
|
||||
t.Helper()
|
||||
var input candidateInput
|
||||
if err := json.Unmarshal(materials.Candidates.Content, &input); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return input.Candidates
|
||||
}
|
||||
|
||||
func cloneCandidates(input []Candidate) []Candidate {
|
||||
output := make([]Candidate, len(input))
|
||||
copy(output, input)
|
||||
for index := range output {
|
||||
output[index].SourceRefs = append([]source.SourceRef(nil), input[index].SourceRefs...)
|
||||
}
|
||||
return output
|
||||
}
|
||||
|
||||
func hasIssue(issues []Issue, want string) bool {
|
||||
for _, issue := range issues {
|
||||
if issue.Category == want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func validateSchema(instanceContent, schemaContent []byte) error {
|
||||
instance, err := jsonschema.UnmarshalJSON(bytes.NewReader(instanceContent))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
document, err := jsonschema.UnmarshalJSON(bytes.NewReader(schemaContent))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
compiler := jsonschema.NewCompiler()
|
||||
if err := compiler.AddResource("schema.json", document); err != nil {
|
||||
return err
|
||||
}
|
||||
compiled, err := compiler.Compile("schema.json")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return compiled.Validate(instance)
|
||||
}
|
||||
@@ -1,175 +0,0 @@
|
||||
package entityreconcile
|
||||
|
||||
import (
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ProposalResponse is the private structured response exchanged with the
|
||||
// reconciliation prompt. It identifies candidates by contextual selectors.
|
||||
type ProposalResponse struct {
|
||||
DuplicateGroups []DuplicateGroup `json:"duplicate_groups"`
|
||||
}
|
||||
|
||||
// DuplicateGroup proposes contextual candidate descriptors that might denote
|
||||
// one entity.
|
||||
type DuplicateGroup struct {
|
||||
Members []Selector `json:"members"`
|
||||
Canonical Selector `json:"canonical"`
|
||||
}
|
||||
|
||||
// Issue identifies one unsafe proposal category without prescribing a warning
|
||||
// message or retry policy to a consuming normalizer.
|
||||
type Issue struct {
|
||||
GroupIndex int
|
||||
Category string
|
||||
}
|
||||
|
||||
// SafeGroup identifies one validated, non-overlapping duplicate group.
|
||||
type SafeGroup struct {
|
||||
members []string
|
||||
canonical string
|
||||
}
|
||||
|
||||
// Members returns an owned copy of the group's candidate keys.
|
||||
func (g SafeGroup) Members() []string { return append([]string(nil), g.members...) }
|
||||
|
||||
// Canonical returns the selected canonical candidate key.
|
||||
func (g SafeGroup) Canonical() string { return g.canonical }
|
||||
|
||||
// Assessment contains only safe groups. Its accessors return owned copies so
|
||||
// callers cannot mutate retained assessment data.
|
||||
type Assessment struct {
|
||||
safeGroups []SafeGroup
|
||||
discardedGroups int
|
||||
issues []Issue
|
||||
}
|
||||
|
||||
// SafeGroups returns validated non-overlapping groups in proposal order.
|
||||
func (a Assessment) SafeGroups() []SafeGroup {
|
||||
groups := make([]SafeGroup, len(a.safeGroups))
|
||||
for index, group := range a.safeGroups {
|
||||
groups[index] = SafeGroup{members: append([]string(nil), group.members...), canonical: group.canonical}
|
||||
}
|
||||
return groups
|
||||
}
|
||||
|
||||
// DiscardedGroups returns the number of rejected proposal groups.
|
||||
func (a Assessment) DiscardedGroups() int { return a.discardedGroups }
|
||||
|
||||
// Issues returns the deterministic rejection categories in proposal order.
|
||||
func (a Assessment) Issues() []Issue { return append([]Issue(nil), a.issues...) }
|
||||
|
||||
// Assess resolves contextual descriptors to internal candidate keys, then
|
||||
// validates the proposal without exposing those keys to the model.
|
||||
func (m Materials) Assess(response ProposalResponse) Assessment {
|
||||
groups := make([]assessedGroup, len(response.DuplicateGroups))
|
||||
issues := make([]Issue, 0)
|
||||
for groupIndex, proposal := range response.DuplicateGroups {
|
||||
groups[groupIndex] = m.assessGroup(proposal)
|
||||
for _, category := range groups[groupIndex].issues {
|
||||
issues = append(issues, Issue{GroupIndex: groupIndex, Category: category})
|
||||
}
|
||||
}
|
||||
|
||||
owners := make(map[string][]int)
|
||||
for groupIndex, group := range groups {
|
||||
if !group.locallyValid {
|
||||
continue
|
||||
}
|
||||
for _, key := range group.members {
|
||||
owners[key] = append(owners[key], groupIndex)
|
||||
}
|
||||
}
|
||||
for groupIndex := range groups {
|
||||
if !groups[groupIndex].locallyValid {
|
||||
continue
|
||||
}
|
||||
for _, key := range groups[groupIndex].members {
|
||||
if len(owners[key]) > 1 {
|
||||
groups[groupIndex].conflicting = true
|
||||
issues = append(issues, Issue{GroupIndex: groupIndex, Category: "overlapping_member"})
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
assessment := Assessment{issues: issues}
|
||||
for _, group := range groups {
|
||||
if !group.locallyValid || group.conflicting {
|
||||
assessment.discardedGroups++
|
||||
continue
|
||||
}
|
||||
assessment.safeGroups = append(assessment.safeGroups, SafeGroup{members: append([]string(nil), group.members...), canonical: group.canonical})
|
||||
}
|
||||
return assessment
|
||||
}
|
||||
|
||||
type assessedGroup struct {
|
||||
members []string
|
||||
canonical string
|
||||
issues []string
|
||||
locallyValid bool
|
||||
conflicting bool
|
||||
}
|
||||
|
||||
func (m Materials) assessGroup(proposal DuplicateGroup) assessedGroup {
|
||||
issues := make([]string, 0)
|
||||
members := make([]string, 0, len(proposal.Members))
|
||||
seen := make(map[string]struct{}, len(proposal.Members))
|
||||
for _, selector := range proposal.Members {
|
||||
key, category := m.selectorKey(selector)
|
||||
if category != "" {
|
||||
issues = append(issues, "member_"+category)
|
||||
continue
|
||||
}
|
||||
if _, exists := seen[key]; exists {
|
||||
issues = append(issues, "repeated_member")
|
||||
continue
|
||||
}
|
||||
seen[key] = struct{}{}
|
||||
members = append(members, key)
|
||||
}
|
||||
canonical, canonicalCategory := m.selectorKey(proposal.Canonical)
|
||||
if canonicalCategory != "" {
|
||||
issues = append(issues, "canonical_"+canonicalCategory)
|
||||
}
|
||||
if len(members) < 2 {
|
||||
issues = append(issues, "fewer_than_two_members")
|
||||
}
|
||||
if canonicalCategory == "" && !contains(members, canonical) {
|
||||
issues = append(issues, "canonical_not_member")
|
||||
}
|
||||
sort.Strings(members)
|
||||
return assessedGroup{members: members, canonical: canonical, issues: issues, locallyValid: len(issues) == 0}
|
||||
}
|
||||
|
||||
func (m Materials) selectorKey(selector Selector) (string, string) {
|
||||
if strings.TrimSpace(selector.Name) == "" {
|
||||
return "", "blank"
|
||||
}
|
||||
lookupKey, err := selectorLookupKey(selector)
|
||||
if err != nil {
|
||||
return "", "unknown"
|
||||
}
|
||||
if _, collided := m.collidedSelectors[lookupKey]; collided {
|
||||
return "", "ineligible"
|
||||
}
|
||||
key, ok := m.keyBySelector[lookupKey]
|
||||
if !ok {
|
||||
return "", "unknown"
|
||||
}
|
||||
if _, eligible := m.eligible[key]; !eligible {
|
||||
return "", "ineligible"
|
||||
}
|
||||
return key, ""
|
||||
}
|
||||
|
||||
func contains(values []string, want string) bool {
|
||||
for _, value := range values {
|
||||
if value == want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -1,55 +0,0 @@
|
||||
package entityreconcile
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"io/fs"
|
||||
|
||||
rootassets "gitea.maximumdirect.net/eric/notarius/assets"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
)
|
||||
|
||||
const (
|
||||
ResponseSchemaKey = llm.ResponseSchemaKey("dnd_entity_reconcile_llm")
|
||||
ResponseSchemaID = "notarius.dnd.entity_reconcile.llm"
|
||||
ResponseSchemaName = "notarius_dnd_entity_reconcile_llm_v1"
|
||||
SchemaVersion = "v1"
|
||||
SchemaAssetPath = "schemas/dnd_entity_reconcile_llm.v1.json"
|
||||
)
|
||||
|
||||
func schemaAssetFS() (fs.FS, error) {
|
||||
assets, err := fs.Sub(rootassets.FS(), "dnd/entity-reconciliation")
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("scope entity reconciliation assets: %w", err)
|
||||
}
|
||||
return assets, nil
|
||||
}
|
||||
|
||||
// LoadResponseSchema returns the shared private duplicate-group response
|
||||
// contract. It is intentionally separate from durable artifact schemas.
|
||||
func LoadResponseSchema() (llm.ResponseSchema, error) {
|
||||
assets, err := schemaAssetFS()
|
||||
if err != nil {
|
||||
return llm.ResponseSchema{}, err
|
||||
}
|
||||
return llm.LoadResponseSchema(assets, llm.ResponseSchemaDefinition{
|
||||
Key: ResponseSchemaKey,
|
||||
ID: ResponseSchemaID,
|
||||
Version: SchemaVersion,
|
||||
Name: ResponseSchemaName,
|
||||
AssetPath: SchemaAssetPath,
|
||||
})
|
||||
}
|
||||
|
||||
// RegisterSchemaAssets makes the shared private response schema available to
|
||||
// prompt preparation. A family registrar can register it once for all consumers.
|
||||
func RegisterSchemaAssets(registry *llm.AssetRegistry) error {
|
||||
assets, err := schemaAssetFS()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
schemas, err := fs.Sub(assets, "schemas")
|
||||
if err != nil {
|
||||
return fmt.Errorf("scope entity reconciliation schemas: %w", err)
|
||||
}
|
||||
return registry.RegisterSchemaFS(schemas, ".")
|
||||
}
|
||||
@@ -353,8 +353,8 @@ func TestEncodeIncludesValidatedEvidenceContext(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("Decode(evidence context file) error = %v", err)
|
||||
}
|
||||
if len(value.Contexts) != 0 {
|
||||
t.Fatalf("evidence context = %#v, want explicit empty contexts", value)
|
||||
if len(value) != 1 || value[0].ID != 7 || value[0].Text != "Source content retained only in the evidence artifact." {
|
||||
t.Fatalf("evidence context = %#v, want the published source-unit array", value)
|
||||
}
|
||||
index := decodeObject(t, fileBytes(t, result.Files, "index.json"))
|
||||
if got, want := index["evidence_context"], map[string]any{
|
||||
@@ -786,7 +786,8 @@ func acceptedEvidenceContextArtifact(t *testing.T) contracts.SerializedArtifact
|
||||
}
|
||||
document.Digest = digest
|
||||
artifact, err := evidencecontext.Serialize(evidencecontext.BuildRequest{
|
||||
Source: document, WindowUnits: 3, SelectedLanes: []string{"spells"},
|
||||
Source: document, WindowUnits: 3,
|
||||
SourceRefs: []source.SourceRef{{SourceID: "source-1", StartUnitID: 7, EndUnitID: 7}},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
|
||||
@@ -6,6 +6,7 @@ import (
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/generic/chunk/units"
|
||||
jsonoutput "gitea.maximumdirect.net/eric/notarius/internal/modules/generic/output/json"
|
||||
alwaysaccept "gitea.maximumdirect.net/eric/notarius/internal/modules/generic/validate/always_accept"
|
||||
@@ -16,14 +17,17 @@ import (
|
||||
|
||||
// Register adds all production domain-neutral modules and validators.
|
||||
func Register(registries pipeline.Registries, assets *llm.AssetRegistry) error {
|
||||
_ = assets
|
||||
if err := validateRegistries(registries); err != nil {
|
||||
return err
|
||||
}
|
||||
if assets == nil {
|
||||
return fmt.Errorf("generic registrar: asset registry must not be nil")
|
||||
}
|
||||
registrations := []struct {
|
||||
name string
|
||||
register func() error
|
||||
}{
|
||||
{name: "semantic reconciliation assets", register: func() error { return semanticreconcile.RegisterAssets(assets) }},
|
||||
{name: "generic chunker", register: func() error { return units.Register(registries.Chunkers) }},
|
||||
{name: "always accept validator", register: func() error { return alwaysaccept.Register(registries.Validators) }},
|
||||
{name: "always reject validator", register: func() error { return alwaysreject.Register(registries.Validators) }},
|
||||
|
||||
@@ -1,15 +1,18 @@
|
||||
package register
|
||||
|
||||
import (
|
||||
"io/fs"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
)
|
||||
|
||||
func TestRegisterAddsGenericFamily(t *testing.T) {
|
||||
registries := completeRegistries()
|
||||
if err := Register(registries, nil); err != nil {
|
||||
assets := llm.NewAssetRegistry()
|
||||
if err := Register(registries, assets); err != nil {
|
||||
t.Fatalf("Register() error = %v, want nil", err)
|
||||
}
|
||||
assertContainsKeys(t, "chunkers", registries.Chunkers.RegisteredKeys(), []string{"generic"})
|
||||
@@ -30,6 +33,31 @@ func TestRegisterAddsGenericFamily(t *testing.T) {
|
||||
if output, err := registries.Outputs.Build("json"); err != nil || output.Key() != "json" {
|
||||
t.Fatalf("build json output = %v, %v; want json implementation", output, err)
|
||||
}
|
||||
promptAssets, err := assets.PromptFS()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := fs.ReadFile(promptAssets, "generic.semantic_reconciliation/prompt.yaml"); err != nil {
|
||||
t.Fatalf("registered semantic reconciliation prompt: %v", err)
|
||||
}
|
||||
schemaAssets, err := assets.SchemaFS()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, err := fs.ReadFile(schemaAssets, "semantic_reconciliation_llm.v1.json"); err != nil {
|
||||
t.Fatalf("registered semantic reconciliation schema: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegisterRejectsNilAssetRegistryBeforeMutation(t *testing.T) {
|
||||
registries := completeRegistries()
|
||||
err := Register(registries, nil)
|
||||
if err == nil || !strings.Contains(err.Error(), "asset registry must not be nil") {
|
||||
t.Fatalf("Register() error = %v, want nil asset registry error", err)
|
||||
}
|
||||
if len(registries.Chunkers.RegisteredKeys()) != 0 {
|
||||
t.Fatalf("chunker keys = %#v, want validation before mutation", registries.Chunkers.RegisteredKeys())
|
||||
}
|
||||
}
|
||||
|
||||
func TestRegisterRejectsMissingGenericRegistriesBeforeMutation(t *testing.T) {
|
||||
@@ -48,7 +76,7 @@ func TestRegisterRejectsMissingGenericRegistriesBeforeMutation(t *testing.T) {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
registries := completeRegistries()
|
||||
test.remove(®istries)
|
||||
err := Register(registries, nil)
|
||||
err := Register(registries, llm.NewAssetRegistry())
|
||||
if err == nil || !strings.Contains(err.Error(), test.wantErr) {
|
||||
t.Fatalf("Register() error = %v, want %q", err, test.wantErr)
|
||||
}
|
||||
@@ -61,11 +89,12 @@ func TestRegisterRejectsMissingGenericRegistriesBeforeMutation(t *testing.T) {
|
||||
|
||||
func TestRegisterReportsDuplicateGenericRegistration(t *testing.T) {
|
||||
registries := completeRegistries()
|
||||
if err := Register(registries, nil); err != nil {
|
||||
assets := llm.NewAssetRegistry()
|
||||
if err := Register(registries, assets); err != nil {
|
||||
t.Fatalf("first Register() error = %v, want nil", err)
|
||||
}
|
||||
err := Register(registries, nil)
|
||||
if err == nil || !strings.Contains(err.Error(), "register generic chunker") || !strings.Contains(err.Error(), "already registered") {
|
||||
err := Register(registries, assets)
|
||||
if err == nil || !strings.Contains(err.Error(), "register semantic reconciliation assets") || !strings.Contains(err.Error(), "already registered") {
|
||||
t.Fatalf("second Register() error = %v, want contextual duplicate error", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"reflect"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
@@ -71,20 +72,17 @@ func TestLocationRegistryHandoffProducesOccurrencesAndEvidence(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("Decode(evidence context) error = %v", err)
|
||||
}
|
||||
if !locationEvidenceHasLane(evidence, "locations") || !locationEvidenceHasLane(evidence, "occurrences") {
|
||||
t.Fatalf("evidence context = %#v, want registry and occurrence evidence from their own artifacts", evidence)
|
||||
if actual := locationEvidenceUnitIDs(evidence); !reflect.DeepEqual(actual, []int{1, 2, 3, 4, 5}) {
|
||||
t.Fatalf("evidence context = %#v, want deduplicated registry and occurrence source-unit evidence", evidence)
|
||||
}
|
||||
}
|
||||
|
||||
func locationEvidenceHasLane(document evidencecontext.Document, laneID string) bool {
|
||||
for _, context := range document.Contexts {
|
||||
for _, reference := range context.EvidenceRefs {
|
||||
if reference.LaneID == laneID {
|
||||
return true
|
||||
}
|
||||
}
|
||||
func locationEvidenceUnitIDs(document evidencecontext.Document) []int {
|
||||
ids := make([]int, len(document))
|
||||
for index, unit := range document {
|
||||
ids[index] = unit.ID
|
||||
}
|
||||
return false
|
||||
return ids
|
||||
}
|
||||
|
||||
func TestLocationOccurrenceConsumerDoesNotRunAfterRejectedRegistry(t *testing.T) {
|
||||
|
||||
@@ -17,6 +17,7 @@ import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/evidencecontext"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
|
||||
combatcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/combatturns"
|
||||
npccodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/npcregistry"
|
||||
@@ -49,12 +50,12 @@ func TestNPCOutputGroundsSpellAndCombatConsumersThroughOneOperation(t *testing.T
|
||||
t.Fatalf("Prepare() error = %v", err)
|
||||
}
|
||||
for name, value := range map[string]string{
|
||||
"extract:npc_registry:dnd/npc-registry:mapping_policy": "dnd.npc_registry.extract_mapping.v2",
|
||||
"normalize:npc_registry:dnd/npc-registry:identity_policy": "dnd.npc_registry.identity.v1",
|
||||
"normalize:npc_registry:dnd/npc-registry:normalization_policy": "dnd.npc_registry.normalize.v4",
|
||||
"normalize:npc_registry:dnd/npc-registry:semantic_context_policy": "dnd.entity_reconcile.context.v1:2",
|
||||
"extract:spells:dnd/spells:mapping_policy": "dnd.spells.extract_mapping.v2",
|
||||
"extract:combat:dnd/combat-turns:scene_gate_policy": "dnd.combat_turns.scene_gate.v1",
|
||||
"extract:npc_registry:dnd/npc-registry:mapping_policy": "dnd.npc_registry.extract_mapping.v2",
|
||||
"normalize:npc_registry:dnd/npc-registry:identity_policy": "dnd.npc_registry.identity.v1",
|
||||
"normalize:npc_registry:dnd/npc-registry:normalization_policy": npcnormalize.NormalizationPolicy,
|
||||
"normalize:npc_registry:dnd/npc-registry:semantic_reconciliation_policy": semanticreconcile.Policy,
|
||||
"extract:spells:dnd/spells:mapping_policy": "dnd.spells.extract_mapping.v2",
|
||||
"extract:combat:dnd/combat-turns:scene_gate_policy": "dnd.combat_turns.scene_gate.v1",
|
||||
} {
|
||||
assertFingerprintValue(t, prepared.CheckpointFingerprints(), name, value)
|
||||
}
|
||||
@@ -63,6 +64,7 @@ func TestNPCOutputGroundsSpellAndCombatConsumersThroughOneOperation(t *testing.T
|
||||
"extract:npc_registry:dnd/npc-registry:response_schema",
|
||||
"normalize:npc_registry:dnd/npc-registry:prompt",
|
||||
"normalize:npc_registry:dnd/npc-registry:response_schema",
|
||||
"normalize:npc_registry:dnd/npc-registry:semantic_reconciliation_limits",
|
||||
"extract:spells:dnd/spells:prompt",
|
||||
"extract:spells:dnd/spells:response_schema",
|
||||
"extract:spells:dnd/spells:npc_registry",
|
||||
@@ -277,22 +279,8 @@ func TestProductionDNDOutputPublishesSelectedEvidenceContext(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("Decode(evidence context) error = %v", err)
|
||||
}
|
||||
if !reflect.DeepEqual(value.SelectedLanes, []string{"combat", "npc_registry", "spells"}) {
|
||||
t.Fatalf("selected lanes = %#v, want configured production lanes without scene descriptions", value.SelectedLanes)
|
||||
}
|
||||
if len(value.Contexts) != 2 || len(value.Contexts[0].Units) != 1 || len(value.Contexts[1].Units) != 1 || value.Contexts[0].Units[0].ID != 10 || value.Contexts[1].Units[0].ID != 20 {
|
||||
t.Fatalf("evidence contexts = %#v, want source-position union with non-monotonic unit IDs", value.Contexts)
|
||||
}
|
||||
firstRefs := value.Contexts[0].EvidenceRefs
|
||||
if len(firstRefs) != 3 || firstRefs[0].LaneID != "combat" || firstRefs[1].LaneID != "npc_registry" || firstRefs[2].LaneID != "spells" {
|
||||
t.Fatalf("first context evidence = %#v, want overlapping selected lane references", firstRefs)
|
||||
}
|
||||
for _, context := range value.Contexts {
|
||||
for _, reference := range context.EvidenceRefs {
|
||||
if reference.LaneID == "scene-descriptions" {
|
||||
t.Fatalf("evidence refs = %#v, want scene descriptions excluded by allowlist", value.Contexts)
|
||||
}
|
||||
}
|
||||
if actual := evidenceUnitIDs(value); !reflect.DeepEqual(actual, []int{10, 20}) {
|
||||
t.Fatalf("evidence units = %#v, want deduplicated source-position union without scene descriptions", actual)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -310,11 +298,19 @@ func TestProductionDNDOutputCanExplicitlySelectSceneDescriptionEvidence(t *testi
|
||||
if err != nil {
|
||||
t.Fatalf("Decode(evidence context) error = %v", err)
|
||||
}
|
||||
if !reflect.DeepEqual(value.SelectedLanes, []string{"scene-descriptions"}) || len(value.Contexts) == 0 || len(value.Contexts[0].EvidenceRefs) == 0 || value.Contexts[0].EvidenceRefs[0].LaneID != "scene-descriptions" {
|
||||
t.Fatalf("evidence context = %#v, want explicitly selected scene-description evidence", value)
|
||||
if len(value) == 0 {
|
||||
t.Fatalf("evidence context = %#v, want explicitly selected scene-description source units", value)
|
||||
}
|
||||
}
|
||||
|
||||
func evidenceUnitIDs(value evidencecontext.Document) []int {
|
||||
ids := make([]int, len(value))
|
||||
for index, unit := range value {
|
||||
ids[index] = unit.ID
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
func TestGroundedPipelineSkipsCombatForExactNarrativeScene(t *testing.T) {
|
||||
registries := productionNPCRegistries(t)
|
||||
configValue := loadGroundedPipelineConfig(t)
|
||||
|
||||
@@ -292,10 +292,7 @@ func (client *semanticNPCOccurrenceClient) CompleteStructured(_ context.Context,
|
||||
}
|
||||
payload = map[string]any{"npcs": []any{map[string]any{"name": name, "source_refs": []any{map[string]int{"start_unit_id": client.npcCalls, "end_unit_id": client.npcCalls}}}}}
|
||||
case npcnormalize.PromptID:
|
||||
content, err := contextualReconciliationContent([]byte(`{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000001"}]}`), request.Inputs["candidates"].Content)
|
||||
if err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, err
|
||||
}
|
||||
content := []byte(`{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":1}]}`)
|
||||
if err := json.Unmarshal(content, out); err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, err
|
||||
}
|
||||
|
||||
@@ -5,7 +5,6 @@ import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
@@ -13,12 +12,12 @@ import (
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/llm"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/framework/semanticreconcile"
|
||||
npccodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/npcregistry"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/npcregistry"
|
||||
npcnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/npcregistry"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/npcs/identity"
|
||||
dndregister "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/register"
|
||||
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/shared/entityreconcile"
|
||||
npcshape "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/validate/npcregistry/shape"
|
||||
npcregistrysourcerefs "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/validate/npcregistry/source_refs"
|
||||
genericregister "gitea.maximumdirect.net/eric/notarius/internal/modules/generic/register"
|
||||
@@ -98,7 +97,7 @@ func TestRunnerProcessesSeriatimInputWithProductionDNDNPCPipeline(t *testing.T)
|
||||
t.Fatalf("manifest lane = %#v, want NPC production composition", lane)
|
||||
}
|
||||
normalizerMetadata, ok := lane.Metadata["normalizer"].(map[string]any)
|
||||
if !ok || normalizerMetadata["identity_policy"] != identity.Policy || normalizerMetadata["normalization_policy"] != npcnormalize.NormalizationPolicy || normalizerMetadata["prompt_id"] != npcnormalize.PromptID || normalizerMetadata["response_schema_id"] != entityreconcile.ResponseSchemaID {
|
||||
if !ok || normalizerMetadata["identity_policy"] != identity.Policy || normalizerMetadata["normalization_policy"] != npcnormalize.NormalizationPolicy || normalizerMetadata["prompt_id"] != npcnormalize.PromptID || normalizerMetadata["response_schema_id"] != semanticreconcile.ResponseSchemaID {
|
||||
t.Fatalf("normalizer metadata = %#v, want identity and normalization policies", lane.Metadata)
|
||||
}
|
||||
var npcOutputFile *contracts.OutputFile
|
||||
@@ -180,8 +179,8 @@ func TestProductionNPCNormalizationRetryUsesFinalSafeProposal(t *testing.T) {
|
||||
{Name: "Mira Thorn", SourceRefs: []npcProductionSourceRef{{StartUnitID: 2, EndUnitID: 2}}},
|
||||
{Name: "Hooded Guard", SourceRefs: []npcProductionSourceRef{{StartUnitID: 3, EndUnitID: 3}}},
|
||||
}}
|
||||
partial := []byte(`{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000002"},{"members":["candidate-000003","unknown"],"canonical":"candidate-000003"}]}`)
|
||||
safe := []byte(`{"duplicate_groups":[{"members":["candidate-000001","candidate-000002"],"canonical":"candidate-000002"}]}`)
|
||||
partial := []byte(`{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2},{"candidate_ids":[3,99],"canonical_candidate_id":3}]}`)
|
||||
safe := []byte(`{"duplicate_groups":[{"candidate_ids":[1,2],"canonical_candidate_id":2}]}`)
|
||||
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
@@ -267,11 +266,6 @@ func (client *fakeNPCProductionLLMClient) CompleteStructured(_ context.Context,
|
||||
}
|
||||
content = append([]byte(nil), client.normalizeResponses[index]...)
|
||||
}
|
||||
var err error
|
||||
content, err = contextualReconciliationContent(content, req.Inputs["candidates"].Content)
|
||||
if err != nil {
|
||||
return contracts.StructuredCompletionResponse{}, err
|
||||
}
|
||||
default:
|
||||
return contracts.StructuredCompletionResponse{}, fmt.Errorf("unexpected fake NPC prompt %q", req.PromptID)
|
||||
}
|
||||
@@ -281,43 +275,6 @@ func (client *fakeNPCProductionLLMClient) CompleteStructured(_ context.Context,
|
||||
return contracts.StructuredCompletionResponse{Content: content}, nil
|
||||
}
|
||||
|
||||
func contextualReconciliationContent(content, candidateContent []byte) ([]byte, error) {
|
||||
if !strings.Contains(string(content), "candidate-") {
|
||||
return content, nil
|
||||
}
|
||||
var selection struct {
|
||||
DuplicateGroups []struct {
|
||||
Members []string `json:"members"`
|
||||
Canonical string `json:"canonical"`
|
||||
} `json:"duplicate_groups"`
|
||||
}
|
||||
if err := json.Unmarshal(content, &selection); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var candidates struct {
|
||||
Candidates []entityreconcile.Selector `json:"candidates"`
|
||||
}
|
||||
if err := json.Unmarshal(candidateContent, &candidates); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
selector := func(key string) entityreconcile.Selector {
|
||||
index, err := strconv.Atoi(strings.TrimPrefix(key, "candidate-"))
|
||||
if err != nil || index < 1 || index > len(candidates.Candidates) {
|
||||
return entityreconcile.Selector{Name: key, SourceRefs: []entityreconcile.SourceRange{}}
|
||||
}
|
||||
return candidates.Candidates[index-1].Clone()
|
||||
}
|
||||
proposal := entityreconcile.ProposalResponse{DuplicateGroups: make([]entityreconcile.DuplicateGroup, len(selection.DuplicateGroups))}
|
||||
for index, group := range selection.DuplicateGroups {
|
||||
members := make([]entityreconcile.Selector, len(group.Members))
|
||||
for memberIndex, key := range group.Members {
|
||||
members[memberIndex] = selector(key)
|
||||
}
|
||||
proposal.DuplicateGroups[index] = entityreconcile.DuplicateGroup{Members: members, Canonical: selector(group.Canonical)}
|
||||
}
|
||||
return json.Marshal(proposal)
|
||||
}
|
||||
|
||||
func (client *fakeNPCProductionLLMClient) requestCount(promptID string) int {
|
||||
count := 0
|
||||
for _, request := range client.requests {
|
||||
|
||||
Reference in New Issue
Block a user