Compare commits

50 Commits

Author SHA1 Message Date
406ad1d362 Retire the completed asset migration roadmaps 2026-08-05 13:23:00 +00:00
5f207b2ab1 Document centralized LLM asset ownership 2026-08-05 00:49:15 +00:00
c613b306ae Centralize item and enemy event assets 2026-08-05 00:45:04 +00:00
1c7291e17d Centralize spell and combat-turn assets 2026-08-05 00:43:39 +00:00
2345da106a Centralize location LLM assets 2026-08-05 00:42:22 +00:00
03761c97dd Centralize NPC LLM assets 2026-08-05 00:40:47 +00:00
9cb9ee48a7 Centralize scene planning and description assets 2026-08-05 00:38:55 +00:00
08954f17e2 Centralize shared D&D LLM assets 2026-08-05 00:37:19 +00:00
a57f83e30d Centralize generic LLM schema assets 2026-08-05 00:34:56 +00:00
f4c05c34ef Make item event responses compatible with strict schemas 2026-08-04 19:52:47 +00:00
29fcad6e9b Improve D&D registry caching and retire the completed roadmap 2026-08-04 18:29:17 +00:00
f5fd115046 Migrate location registry to shared resolver 2026-08-04 13:30:12 +00:00
7f28899730 Migrate NPC registry to shared resolver 2026-08-04 13:26:06 +00:00
84c0758455 Add shared D&D registry resolver 2026-08-04 13:20:20 +00:00
5002864e88 Narrow location occurrence normalizer references 2026-08-04 13:12:46 +00:00
55b188fd84 Clarify hypothetical location occurrence classification 2026-08-04 13:09:30 +00:00
d1f43df88e Restore NPC canonical name selection 2026-08-04 13:08:00 +00:00
d52387c1f7 Document D&D location tracking contracts 2026-08-04 00:51:43 +00:00
9c5e3cff14 Add D&D location tracking to complete example 2026-08-04 00:45:50 +00:00
811d5b8bd9 Compose D&D location tracking modules 2026-08-04 00:39:00 +00:00
a168c13b85 Add D&D location occurrence validators 2026-08-04 00:33:39 +00:00
dd61a4efda Add D&D location validators 2026-08-04 00:26:50 +00:00
228cc6ee83 Add D&D location occurrence normalizer 2026-08-04 00:22:25 +00:00
06170e1f65 Add D&D location occurrence extractor 2026-08-04 00:18:16 +00:00
7715baa1f6 Add immutable D&D location registry 2026-08-04 00:11:31 +00:00
98506db1a9 Add D&D location normalizer 2026-08-04 00:07:47 +00:00
bb4855f0c6 Add D&D location extractor 2026-08-04 00:02:18 +00:00
c51934d5c6 Migrate NPC normalization to shared reconciliation 2026-08-03 23:56:48 +00:00
c3513da880 Add shared D&D entity reconciliation support 2026-08-03 23:47:42 +00:00
c7d853ea52 Add D&D location artifact codecs 2026-08-03 23:40:35 +00:00
da7fcdaffd Add D&D location identity contracts 2026-08-03 23:35:50 +00:00
b6aad4fa98 Plan D&D location tracking 2026-08-03 23:29:32 +00:00
9c6af28d02 Finish the enemy engagement cleanup 2026-08-03 23:11:44 +00:00
04eabdfcb9 Complete enemy event reference documentation 2026-08-03 23:00:09 +00:00
a8a99c1037 Make enemy event prompt tests resilient to refactoring 2026-08-03 22:57:17 +00:00
e15007fffb Simplify enemy event grounding ownership 2026-08-03 22:54:11 +00:00
e6b7c61f45 Enforce unique enemy engagements per scene 2026-08-03 22:50:41 +00:00
42973215fa Document D&D enemy event artifacts 2026-08-03 21:14:52 +00:00
ba1d112d1f Add D&D enemy event pipeline example 2026-08-03 21:10:21 +00:00
b722131d57 Compose D&D enemy event production family 2026-08-03 21:01:23 +00:00
a92d2c0885 Add D&D enemy event validators 2026-08-03 20:56:36 +00:00
02ec10d66b Add D&D enemy event normalizer 2026-08-03 20:49:32 +00:00
9dd57dbfa4 Add D&D enemy event extractor 2026-08-03 20:44:44 +00:00
c164a3fc69 Prepare D&D enemy event grounding references 2026-08-03 20:37:35 +00:00
8834df617f Add D&D enemy event artifact contract 2026-08-03 20:32:20 +00:00
db2adb52da Preserve prompt sessions and retire completed roadmaps 2026-08-03 19:45:27 +00:00
fc3c128171 Document prompt sessions and concurrency defaults 2026-08-03 19:16:18 +00:00
8a15b083a0 Raise default LLM concurrency 2026-08-03 19:12:19 +00:00
d9dae2b639 Record effective sessions in debug provenance 2026-08-03 19:09:49 +00:00
7a4fd7be7a Derive stable prompt sessions for CLI runs 2026-08-03 19:05:49 +00:00
248 changed files with 14125 additions and 2146 deletions

View File

@@ -2,8 +2,9 @@
Notarius is a Go CLI for turning source material into structured artifacts with
configured extraction pipelines. The implemented D&D workflow reads Seriatim
transcript JSON and can produce scene descriptions, item and currency events,
NPC identities, combat turns, NPC interactions, and spell casts.
transcript JSON and can produce location registries and occurrences, scene
descriptions, item and currency events, NPC identities, combat turns, NPC
interactions, enemy events, and spell casts.
## Quickstart

View File

@@ -0,0 +1,55 @@
id: dnd.enemy_events
version: "v1"
default_profile: dnd-extraction
inputs:
- name: transcript
required: true
content_type: application/json
- name: players
required: false
content_type: text/plain
- name: party
required: false
content_type: text/plain
- name: glossary
required: false
content_type: text/plain
- name: npcs
required: true
content_type: application/json
- name: combat_turns
required: true
content_type: application/json
- name: npc_interactions
required: true
content_type: application/json
messages:
- role: system
content_file: ./sharedassets/common-dnd-system.md
- role: user
content_file: ./sharedassets/common-dnd-identity.md
- role: user
content_file: ./sharedassets/common-dnd-references.md
cache_control:
type: ephemeral
- role: user
content_file: ./sharedassets/common-dnd-transcript.md
cache_control:
type: ephemeral
- role: user
content_file: ./sharedassets/common-dnd-extraction-evidence.md
- role: user
content_file: ./sharedassets/common-dnd-npcs.md
- role: user
content_file: ./grounding.md
- role: user
content_file: ./task.md
- role: user
content_file: ./instructions.md
cache_control:
type: ephemeral
output:
format: json
validation_mode: json_schema
schema_path: dnd_enemy_events_llm.v1.json
repair_attempts: 0

View File

@@ -0,0 +1,12 @@
Compact combat grounding is supplied below. It can guide attention and
disambiguation, but it is not evidence. Do not derive an event, subject,
outcome, or source range from either list. The current transcript alone must
directly establish every returned event.
Combat-turn grounding:
{{ input "combat_turns" }}
Named combat-opponent grounding:
{{ input "npc_interactions" }}

View File

@@ -0,0 +1,14 @@
Exclude party members, allies, neutral observers, mentioned-but-absent enemies,
hazards, traps, environmental effects, uncertain allegiance, table talk,
planning, hypotheses, recaps outside this passage, and downstream inference.
Do not infer an engagement or outcome from initiative, turn absence, damage,
defeat, movement, a scene ending, combat-opponent grounding, or any auxiliary
artifact. Auxiliary inputs can guide attention but cannot prove or supply an
event. Cite only narrow current-transcript ranges that establish each event.
Return the `events` array even when no enemy event is established. Every event
must contain only `name`, `kind`, and `source_refs`. Use exactly one kind:
`engaged`, `killed`, `fled`, `captured`, or `incapacitated`. Each source range
uses integer `start_unit_id` and `end_unit_id`; omit `source_id` because
Notarius assigns the current source identity.

View File

@@ -0,0 +1,20 @@
Extract Dungeons & Dragons enemy events from the supplied combat transcript.
Return an `engaged` event only when the transcript directly establishes that a
subject is actively opposing the party in combat. Return `killed`, `fled`,
`captured`, or `incapacitated` only when the transcript explicitly establishes
that outcome. An outcome may share evidence with an engagement, and a later
engagement or outcome for the same subject remains a separate observation.
Emit at most one engagement for the same subject in this combat scene.
For `killed`, direct death or killing is required. For `fled`, the subject must
explicitly escape, retreat, or leave combat to avoid continued engagement. For
`captured`, the subject must be explicitly taken prisoner or secured under the
party's control. For `incapacitated`, the subject must be explicitly unable to
continue acting without being established as killed or captured.
Use a normalized NPC registry spelling when the transcript identifies that
named NPC. A hostile creature without a registry entry is allowed. For unnamed
individuals or groups, use only the narrowest transcript-grounded label, such
as `Orcs`, `One orc`, or `Remaining orcs`; never invent member names, IDs, or
quantities.

View File

@@ -0,0 +1,33 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "notarius.dnd.enemy_events.llm",
"type": "object",
"additionalProperties": false,
"required": ["events"],
"properties": {
"events": {
"type": "array",
"items": {
"type": "object",
"additionalProperties": false,
"required": ["name", "kind", "source_refs"],
"properties": {
"name": {"type": "string"},
"kind": {"type": "string"},
"source_refs": {
"type": "array",
"items": {
"type": "object",
"additionalProperties": false,
"required": ["start_unit_id", "end_unit_id"],
"properties": {
"start_unit_id": {"type": "integer"},
"end_unit_id": {"type": "integer"}
}
}
}
}
}
}
}
}

View File

@@ -1,6 +1,6 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "notarius.dnd.npcs.normalize.llm",
"$id": "notarius.dnd.entity_reconcile.llm",
"type": "object",
"additionalProperties": false,
"required": ["duplicate_groups"],
@@ -10,13 +10,13 @@
"items": {
"type": "object",
"additionalProperties": false,
"required": ["members", "canonical_name"],
"required": ["members", "canonical"],
"properties": {
"members": {
"type": "array",
"items": {"type": "string"}
},
"canonical_name": {"type": "string"}
"canonical": {"type": "string"}
}
}
}

View File

@@ -1,7 +1,7 @@
Return one event only when the transcript establishes a meaningful item or
currency occurrence. Use a concise observed item name and preserve the stated
currency denomination; include quantity only when the transcript explicitly
states it.
currency denomination. Set `quantity` to the explicitly stated integer, or to
`null` when the transcript does not state one.
Use `discovered` when the party learns of or encounters an item without
establishing possession. Use `acquired` when the party or a party member gains
@@ -13,11 +13,12 @@ when the transcript explicitly describes it being physically destroyed or
expended as a non-payment component. Use `transferred` only when possession
moves between two distinct named party members.
For `discovered`, omit both holders. For `acquired`, provide only `to`; for
`lost` and `consumed`, provide only `from`; and for `transferred`, provide both
`from` and `to`. Use `party` only for collective or unresolved party possession,
never for either side of a transfer. Do not emit a transfer for a gift, sale, or
payment outside the party.
Return both `from` and `to` for every event, using `null` when a holder does not
apply. For `discovered`, set both holders to `null`. For `acquired`, set `from`
to `null` and provide `to`; for `lost` and `consumed`, provide `from` and set
`to` to `null`; and for `transferred`, provide both holders. Use `party` only
for collective or unresolved party possession, never for either side of a
transfer. Do not emit a transfer for a gift, sale, or payment outside the party.
Ordinary non-depleting use is not an event. Do not infer acquisition from a
discovery, or discovery from an acquisition: emit both only when each is

View File

@@ -10,13 +10,13 @@
"items": {
"type": "object",
"additionalProperties": false,
"required": ["name", "kind", "source_refs"],
"required": ["name", "kind", "quantity", "from", "to", "source_refs"],
"properties": {
"name": {"type": "string"},
"kind": {"type": "string"},
"quantity": {"type": "integer"},
"from": {"type": "string"},
"to": {"type": "string"},
"quantity": {"type": ["integer", "null"]},
"from": {"type": ["string", "null"]},
"to": {"type": ["string", "null"]},
"source_refs": {
"type": "array",
"items": {

View File

@@ -0,0 +1,47 @@
id: dnd.location_occurrences
version: "v1"
default_profile: dnd-extraction
inputs:
- name: transcript
required: true
content_type: application/json
- name: players
required: false
content_type: text/plain
- name: party
required: false
content_type: text/plain
- name: glossary
required: false
content_type: text/plain
- name: locations
required: true
content_type: application/json
messages:
- role: system
content_file: ./sharedassets/common-dnd-system.md
- role: user
content_file: ./sharedassets/common-dnd-identity.md
- role: user
content_file: ./sharedassets/common-dnd-references.md
cache_control:
type: ephemeral
- role: user
content_file: ./sharedassets/common-dnd-transcript.md
cache_control:
type: ephemeral
- role: user
content_file: ./sharedassets/common-dnd-extraction-evidence.md
- role: user
content_file: ./locations.md
- role: user
content_file: ./task.md
- role: user
content_file: ./instructions.md
cache_control:
type: ephemeral
output:
format: json
validation_mode: json_schema
schema_path: dnd_location_occurrences_llm.v1.json
repair_attempts: 0

View File

@@ -0,0 +1,8 @@
Return the occurrences array even when no occurrence is established. Every
record must contain location_id, name, kind, and source_refs. Copy location_id
and name from one supplied registry record, and cite only narrow transcript
ranges that support both that location and its classified occurrence.
Do not summarize location descriptions, infer a missing registry record, or
use registry context as evidence. Omit source_id; Notarius assigns the current
transcript source identity.

View File

@@ -0,0 +1,9 @@
A normalized location registry is provided below for identity grounding. It may
be empty. Each record contains the exact location ID and canonical display name
to copy when the transcript establishes an occurrence of that place.
Registry content is context, not occurrence evidence. Do not derive an
occurrence or a source range from the registry, and do not infer a location
that is absent from it.
{{ input "locations" }}

View File

@@ -0,0 +1,29 @@
Extract Dungeons & Dragons location occurrences from the supplied transcript.
Include an occurrence only when the transcript establishes one supplied
location, one occurrence kind, and a coherent passage supporting both. Use
only the exact ID and name pair from the supplied location registry. Return an
empty occurrences array when no supplied location has an evidenced occurrence
in this transcript passage.
Use exactly one kind per occurrence:
- visited: party members are physically present, arrive, remain, or depart;
- planned: the party explicitly proposes, intends, or agrees to future travel;
- recalled: the transcript explicitly recounts an earlier party visit; or
- mentioned: the location is explicitly referenced without stronger support,
including non-actionable speculation or a mere hypothetical reference.
A mere hypothetical or speculative reference is not planned unless the
transcript also establishes an actual proposal, intention, or agreement to
travel. When the hypothetical itself explicitly names a supplied registry
location, it may be mentioned using the narrow passage that supports that
reference.
For overlapping support, visited outranks planned, recalled, and mentioned;
planned outranks recalled and mentioned; recalled outranks mentioned. A passage
may produce multiple records when it independently establishes separate facts,
such as recalling an earlier visit while planning a return. Omit inferred,
unstated, uncertain, or unsupported places and occurrences. Do not infer a
location or occurrence from surrounding events when the transcript does not
state it.

View File

@@ -0,0 +1,34 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "notarius.dnd.location_occurrences.llm",
"type": "object",
"additionalProperties": false,
"required": ["occurrences"],
"properties": {
"occurrences": {
"type": "array",
"items": {
"type": "object",
"additionalProperties": false,
"required": ["location_id", "name", "kind", "source_refs"],
"properties": {
"location_id": {"type": "string"},
"name": {"type": "string"},
"kind": {"enum": ["visited", "planned", "recalled", "mentioned"]},
"source_refs": {
"type": "array",
"items": {
"type": "object",
"additionalProperties": false,
"required": ["start_unit_id", "end_unit_id"],
"properties": {
"start_unit_id": {"type": "integer"},
"end_unit_id": {"type": "integer"}
}
}
}
}
}
}
}
}

View File

@@ -0,0 +1,42 @@
id: dnd.locations
version: "v1"
default_profile: dnd-extraction
inputs:
- name: transcript
required: true
content_type: application/json
- name: players
required: false
content_type: text/plain
- name: party
required: false
content_type: text/plain
- name: glossary
required: false
content_type: text/plain
messages:
- role: system
content_file: ./sharedassets/common-dnd-system.md
- role: user
content_file: ./sharedassets/common-dnd-identity.md
- role: user
content_file: ./sharedassets/common-dnd-references.md
cache_control:
type: ephemeral
- role: user
content_file: ./sharedassets/common-dnd-transcript.md
cache_control:
type: ephemeral
- role: user
content_file: ./sharedassets/common-dnd-extraction-evidence.md
- role: user
content_file: ./task.md
- role: user
content_file: ./instructions.md
cache_control:
type: ephemeral
output:
format: json
validation_mode: json_schema
schema_path: dnd_locations_llm.v1.json
repair_attempts: 0

View File

@@ -0,0 +1,6 @@
Return only observed location display names and narrow transcript source ranges.
Exclude people, creatures, objects, organizations, abstract concepts, and
places merely inferred from an event. Omit uncertain or unsupported places.
Campaign references may clarify terms already present in the transcript, but
they are not evidence and must never supply a source range.

View File

@@ -0,0 +1,8 @@
Extract physical places established by the provided Dungeons & Dragons
transcript and cite where each place is identified.
Include planes, regions, settlements, districts, buildings, rooms, landmarks,
routes, and geographic features. A generic label such as "the tavern" is
allowed only when the transcript uses it for a specific place. Keep aliases and
nested places when the transcript identifies them; do not merge or invent
qualifiers for similarly named places.

View File

@@ -0,0 +1,32 @@
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "notarius.dnd.locations.llm",
"type": "object",
"additionalProperties": false,
"required": ["locations"],
"properties": {
"locations": {
"type": "array",
"items": {
"type": "object",
"additionalProperties": false,
"required": ["name", "source_refs"],
"properties": {
"name": {"type": "string"},
"source_refs": {
"type": "array",
"items": {
"type": "object",
"additionalProperties": false,
"required": ["start_unit_id", "end_unit_id"],
"properties": {
"start_unit_id": {"type": "integer"},
"end_unit_id": {"type": "integer"}
}
}
}
}
}
}
}
}

View File

@@ -0,0 +1,2 @@
Location candidates:
{{ input "candidates" }}

View File

@@ -0,0 +1,30 @@
id: dnd.locations.normalize
version: "v1"
default_profile: dnd-extraction
inputs:
- name: candidates
required: true
content_type: application/json
- name: transcript
required: true
content_type: application/json
messages:
- role: system
content_file: ./sharedassets/common-dnd-system.md
- role: user
content_file: ./task.md
- role: user
content_file: ./sharedassets/common-dnd-entity-reconciliation.md
cache_control:
type: ephemeral
- role: user
content_file: ./candidates.md
- role: user
content_file: ./sharedassets/common-dnd-transcript.md
cache_control:
type: ephemeral
output:
format: json
validation_mode: json_schema
schema_path: dnd_entity_reconcile_llm.v1.json
repair_attempts: 0

View File

@@ -0,0 +1,8 @@
Review location candidates and cited transcript context. Group candidates only
when the evidence clearly identifies one physical place.
Do not group candidates solely because their names match, their evidence is
nearby, one place is nested inside another, or their labels are generic. Keep
parent and child places, similarly named places, and uncertain aliases
separate. For an accepted group, select the supplied candidate with the
clearest established display name as canonical.

View File

@@ -14,7 +14,7 @@ messages:
- role: user
content_file: ./task.md
- role: user
content_file: ./instructions.md
content_file: ./sharedassets/common-dnd-entity-reconciliation.md
cache_control:
type: ephemeral
- role: user
@@ -26,5 +26,5 @@ messages:
output:
format: json
validation_mode: json_schema
schema_path: dnd_npcs_normalize_llm.v1.json
schema_path: dnd_entity_reconcile_llm.v1.json
repair_attempts: 0

View File

@@ -0,0 +1,13 @@
Review NPC candidates and their cited transcript context to identify aliases
that refer to the same individual. Propose only groups supported by the
transcript, and preserve distinct individuals even when their names are
similar.
For every accepted group, choose as canonical only a supplied candidate from
that evidence-supported duplicate group. Prefer a complete, stable proper name
over an abbreviation. Prefer an unadorned proper name over that name plus a
contextual class, role, title, or relationship descriptor unless the transcript
establishes the descriptor as part of the person's name. A longer display name
is not inherently more canonical; for example, do not prefer `Captain Aria`
over `Aria` solely because it includes the contextual title `Captain`. Do not
invent, edit, or combine display names.

View File

@@ -0,0 +1,6 @@
Identify only well-supported duplicate groups among the supplied candidates.
Candidate keys are opaque identifiers. Copy each selected key exactly. A group
must contain at least two supplied keys, and its `canonical` key must be one of
its members. Do not create keys, records, names, source references, evidence,
or replacement values. Omit any uncertain or unsafe group.

15
assets/package.go Normal file
View File

@@ -0,0 +1,15 @@
// Package assets exposes embedded LLM-facing content.
package assets
import (
"embed"
"io/fs"
)
//go:embed dnd generic
var embedded embed.FS
// FS returns the embedded read-only asset filesystem.
func FS() fs.FS {
return embedded
}

View File

@@ -1,6 +1,6 @@
# ADR-0004: Package modules by domain, not by stage
**Status:** Accepted
**Status:** Accepted — its asset-co-location rule is superseded by [ADR-0011](0011-centralize-llm-assets.md); its domain-first module packaging decision remains accepted.
**Date:** 2026-07-13
## Context

View File

@@ -0,0 +1,69 @@
# ADR-0011: Centralize LLM-facing assets in a content-only package
**Status:** Accepted
**Date:** 2026-08-05
## Context
LLM prompts, private response schemas, generic schemas, and fallback profiles
are authored and reviewed as content, but package-local embedding scattered that
content across implementation trees. Finding all of the assets that contribute
to a prompt family required navigating code ownership boundaries rather than a
single discoverable content boundary.
The repository must retain module ownership of prompt semantics, schema
identities, registration, and prompt-cache behavior. Durable artifact schemas
and non-LLM domain data have different compatibility and ownership rules, so
they must not move merely because they are embedded files.
## Decision
LLM-facing content is embedded by the root `assets` package. It is a data-only
dependency leaf: its single `FS() fs.FS` API returns the read-only embedded
filesystem, and the package contains no business logic or internal or PromptKit
dependencies. The accepted import path is
`gitea.maximumdirect.net/eric/notarius/assets`; it makes repository-owned
content available to its consumers, not a public extension contract.
Consumers scope that filesystem to the subtree they own before reading or
registering content. Modules continue to own their manifests, prompt ordering,
private response-schema identity, and registration. Centralizing physical files
does not centralize domain semantics or transfer those responsibilities to the
root package.
The root package contains prompt content, private LLM response schemas, generic
LLM schemas, shared fragments, and fallback profiles. Durable artifact schemas
and non-LLM domain data remain with their current owners. A module fingerprint
is derived from its manifest-selected module and shared files, rather than from
an entire asset tree. The relocation is accepted to cause a one-time checkpoint
invalidation.
This decision supersedes only the physical asset-co-location portion of
ADR-0004's decision that places domain-specific prompt fragments and schemas
within the domain tree. ADR-0004's domain-first packaging and registrar
ownership decisions remain accepted.
## Alternatives Considered
- Keep package-local assets. This preserves physical co-location with code but
makes prompt-author discovery and cross-family review unnecessarily costly.
- Use `internal/llmassets`. This would hide content from legitimate owners
outside the `internal` subtree and would make the root asset boundary depend
on implementation-layer placement.
- Build a behavioral central registry. This would mix content discovery with
prompt selection and registration behavior, moving module semantics into a
shared registry.
- Use runtime filesystem overlays. This would add runtime configuration and
failure modes where compile-time embedded content is sufficient.
## Consequences
Prompt authors can find in-scope LLM content in one top-level tree while module
packages continue to define its meaning and registration. Consumers have an
explicit, narrow dependency on only the content they need. The root package is
intentionally importable but must remain a stable, content-only leaf rather
than becoming a general extension API.
The initial relocation invalidates existing checkpoints once. Later checkpoint
identity changes remain limited to the manifest-selected prompt and shared
content, so unrelated files do not trigger recomputation.

View File

@@ -40,7 +40,7 @@ pipeline ID and **--input** are required.
| **--debug-dir path** | Override the debug-bundle root. Requires **--debug**. |
| **--only lane-a,lane-b** | Run only the selected comma-separated artifact lanes when that selection is valid for the configured pipeline. |
| **--llm-profile id** | Highest-precedence configured profile for selected LLM-backed bindings and validators; it replaces binding and [pipeline](config.md#pipelines) defaults. |
| **--session-id id** | Supply a non-empty prompt session identifier to LLM-backed module calls. |
| **--session-id id** | Override the generated prompt session identifier with a non-empty value for LLM-backed module calls. |
| **--reasoning-effort value** | Replace the selected PromptKit profile's reasoning effort for every LLM-backed call in this run. The value must be non-empty and the flag may be specified only once. |
| **--clear-reasoning-effort** | Clear reasoning effort inherited from the selected PromptKit profile for every LLM-backed call in this run. |
| **--reference selector=path** | Add or replace a file reference binding. Repeatable. |
@@ -57,6 +57,16 @@ Persistent reasoning settings remain a PromptKit profile concern.
**--recompute-step** requires **--resume**; checkpoint requirements and reuse
behavior are documented in [Operations](operations.md).
Every run uses one effective prompt session. Without **--session-id**, Notarius
generates a stable `notarius:v1:` identifier from the trimmed resolved input
module key and the input file's exact raw bytes. The same module and bytes
therefore produce the same identifier, regardless of pipeline, references,
profile, retries, or run settings. An explicit non-empty value replaces that
default. Session identifiers are visible to providers; they are non-secret
correlation identifiers, not credential storage. See
[Operations](operations.md#operational-limits) for privacy and workflow
guidance.
### Reference selectors
Use **--reference** only for a reference slot declared by the selected

View File

@@ -58,7 +58,7 @@ Built-in defaults are:
| Field | Default |
| --- | --- |
| **concurrency.total_llm** | 1 |
| **concurrency.total_llm** | 16 |
| **concurrency.stage_workers.extract** | Effective **total_llm** |
| **output.directory** | **./notarius-output** |
| **cache.chunk_plans.mode** | **auto** |
@@ -370,14 +370,33 @@ selected target declares them:
| **players** | Optional text player context. |
| **glossary** | Optional text campaign glossary. |
| **spell_catalog** | Optional JSON spell-catalog overlay for spell extraction and normalization. See [spell-catalog overlays](integrations/dnd-spell-catalog-overlays.md). |
| **npcs** | Normalized NPC registry. Optional for spells and combat turns; required for NPC interactions. |
| **scene_descriptions** | Required normalized scene-description artifact for combat-turn extraction. |
| **locations** | Required normalized location registry for location-occurrence extraction and normalization. |
| **npcs** | Normalized NPC registry. Optional for spells and combat turns; required for NPC interactions and enemy-event extraction and normalization. |
| **scene_descriptions** | Required normalized scene-description artifact for combat-turn and enemy-event extraction. |
| **combat_turns** | Required normalized combat-turn artifact for enemy-event extraction. |
| **npc_interactions** | Required normalized NPC-interaction artifact for enemy-event extraction. |
Location-occurrence and enemy-event artifact slots have the following exact
binding contracts. Durable semantics and wire shapes remain in their
[location-occurrence](integrations/dnd-location-occurrence-artifacts.md) and
[enemy-event](integrations/dnd-enemy-event-artifacts.md) contracts.
| Slot | Accepted artifact kind | Media type | Maximum size | Required stage |
| --- | --- | --- | --- | --- |
| `npcs` | `dnd/npc-list` | `application/json` | 1,048,576 bytes | extract and normalize |
| `scene_descriptions` | `dnd/scene-description-list` | `application/json` | 1,048,576 bytes | extract only |
| `combat_turns` | `dnd/combat-turn-list` | `application/json` | 1,048,576 bytes | extract only |
| `npc_interactions` | `dnd/npc-interaction-list` | `application/json` | 1,048,576 bytes | extract only |
| `locations` | `dnd/location-list` | `application/json` | 1,048,576 bytes | location-occurrence extract and normalize |
Scene descriptions accept **party**, **players**, and **glossary**, but not
**roster**. NPC interactions require **npcs** for both extraction and
normalization. Combat turns require **scene_descriptions** for extraction; the
normalized combat-turn module may use optional **npcs**. The complete example
shows generated **npcs** and **scene_descriptions** bindings.
normalized combat-turn module may use optional **npcs**. Location occurrences
require **locations** for extraction and normalization. Enemy-event extraction
requires all four of its JSON artifact slots; its normalizer requires **npcs**.
The [complete example](../examples/dnd-complete.config.yml) shows the ordered
generated bindings.
## Production Module Keys
@@ -385,18 +404,27 @@ shows generated **npcs** and **scene_descriptions** bindings.
| --- | --- |
| Input | **seriatim** |
| Chunk | **generic**, **dnd/scenes** |
| Extract | **dnd/spells**, **dnd/npcs**, **dnd/combat-turns**, **dnd/item-events**, **dnd/npc-interactions**, **dnd/scene-descriptions** |
| Extract | **dnd/spells**, **dnd/npcs**, **dnd/combat-turns**, **dnd/item-events**, **dnd/npc-interactions**, **dnd/scene-descriptions**, **dnd/enemy-events**, **dnd/locations**, **dnd/location-occurrences** |
| Merge | **appendorder** |
| Normalize | **noop**, **dnd/spells**, **dnd/npcs**, **dnd/combat-turns**, **dnd/item-events**, **dnd/npc-interactions**, **dnd/scene-descriptions** |
| Normalize | **noop**, **dnd/spells**, **dnd/npcs**, **dnd/combat-turns**, **dnd/item-events**, **dnd/npc-interactions**, **dnd/scene-descriptions**, **dnd/enemy-events**, **dnd/locations**, **dnd/location-occurrences** |
| Output | **json** |
`dnd/locations` extraction and normalization are `llm_backed`; location
normalization may use the pipeline's selected LLM profile for bounded duplicate
proposals. `dnd/location-occurrences` extraction is `llm_backed`, while its
normalizer is `deterministic`. The complete example binds the registry in one
step and the occurrence lane in the next.
The D&D artifact contracts define each emitted schema:
[spells](integrations/dnd-spell-artifacts.md),
[NPCs](integrations/dnd-npc-artifacts.md),
[NPC interactions](integrations/dnd-npc-interaction-artifacts.md),
[combat turns](integrations/dnd-combat-turn-artifacts.md),
[item events](integrations/dnd-item-event-artifacts.md), and
[scene descriptions](integrations/dnd-scene-description-artifacts.md).
[item events](integrations/dnd-item-event-artifacts.md),
[scene descriptions](integrations/dnd-scene-description-artifacts.md), and
[enemy events](integrations/dnd-enemy-event-artifacts.md),
[locations](integrations/dnd-location-artifacts.md), and
[location occurrences](integrations/dnd-location-occurrence-artifacts.md).
## Production Validator Keys And Default Chains
@@ -411,6 +439,9 @@ Available validator keys are:
| Item events | **extract/dnd/item-events/shape**, **extract/dnd/item-events/source_refs**, **extract/dnd/item-events/source_relatedness**, **normalize/dnd/item-events/invariants** |
| NPC interactions | **extract/dnd/npc-interactions/shape**, **extract/dnd/npc-interactions/registry**, **extract/dnd/npc-interactions/source_refs**, **extract/dnd/npc-interactions/source_relatedness**, **normalize/dnd/npc-interactions/invariants** |
| Scene descriptions | **extract/dnd/scene-descriptions/shape**, **extract/dnd/scene-descriptions/source_refs**, **extract/dnd/scene-descriptions/source_relatedness**, **normalize/dnd/scene-descriptions/invariants** |
| Enemy events | **extract/dnd/enemy-events/shape**, **extract/dnd/enemy-events/engagements**, **extract/dnd/enemy-events/source_refs**, **extract/dnd/enemy-events/source_relatedness**, **normalize/dnd/enemy-events/invariants** |
| Locations | **extract/dnd/locations/shape**, **extract/dnd/locations/source_refs**, **extract/dnd/locations/source_relatedness**, **normalize/dnd/locations/identity** |
| Location occurrences | **extract/dnd/location-occurrences/shape**, **extract/dnd/location-occurrences/registry**, **extract/dnd/location-occurrences/source_refs**, **extract/dnd/location-occurrences/source_relatedness**, **normalize/dnd/location-occurrences/invariants** |
When no override is configured, production D&D bindings use the following
ordered chains. Each row lists extract then normalize; spell chains are the
@@ -424,6 +455,9 @@ same at both stages.
| Item events | generic/valid_json, extract/dnd/item-events/shape, extract/dnd/item-events/source_refs, generic/valid_json_schema, extract/dnd/item-events/source_relatedness | generic/valid_json, extract/dnd/item-events/shape, normalize/dnd/item-events/invariants, extract/dnd/item-events/source_refs, generic/valid_json_schema, extract/dnd/item-events/source_relatedness |
| NPC interactions | generic/valid_json, extract/dnd/npc-interactions/shape, extract/dnd/npc-interactions/registry, extract/dnd/npc-interactions/source_refs, generic/valid_json_schema, extract/dnd/npc-interactions/source_relatedness | generic/valid_json, extract/dnd/npc-interactions/shape, extract/dnd/npc-interactions/registry, normalize/dnd/npc-interactions/invariants, extract/dnd/npc-interactions/source_refs, generic/valid_json_schema, extract/dnd/npc-interactions/source_relatedness |
| Scene descriptions | generic/valid_json, extract/dnd/scene-descriptions/shape, extract/dnd/scene-descriptions/source_refs, generic/valid_json_schema, extract/dnd/scene-descriptions/source_relatedness | generic/valid_json, extract/dnd/scene-descriptions/shape, normalize/dnd/scene-descriptions/invariants, extract/dnd/scene-descriptions/source_refs, generic/valid_json_schema, extract/dnd/scene-descriptions/source_relatedness |
| Enemy events | generic/valid_json, extract/dnd/enemy-events/shape, extract/dnd/enemy-events/engagements, extract/dnd/enemy-events/source_refs, generic/valid_json_schema, extract/dnd/enemy-events/source_relatedness | generic/valid_json, extract/dnd/enemy-events/shape, normalize/dnd/enemy-events/invariants, extract/dnd/enemy-events/source_refs, generic/valid_json_schema, extract/dnd/enemy-events/source_relatedness |
| Locations | generic/valid_json, extract/dnd/locations/shape, extract/dnd/locations/source_refs, generic/valid_json_schema, extract/dnd/locations/source_relatedness | generic/valid_json, extract/dnd/locations/shape, normalize/dnd/locations/identity, extract/dnd/locations/source_refs, generic/valid_json_schema, extract/dnd/locations/source_relatedness |
| Location occurrences | generic/valid_json, extract/dnd/location-occurrences/shape, extract/dnd/location-occurrences/registry, extract/dnd/location-occurrences/source_refs, generic/valid_json_schema, extract/dnd/location-occurrences/source_relatedness | generic/valid_json, extract/dnd/location-occurrences/shape, extract/dnd/location-occurrences/registry, normalize/dnd/location-occurrences/invariants, extract/dnd/location-occurrences/source_refs, generic/valid_json_schema, extract/dnd/location-occurrences/source_relatedness |
Chains are only registered for the D&D extract and normalize modules shown
above; select an explicit override when a different compatible chain is

View File

@@ -27,10 +27,13 @@ notarius run pipeline-id \
```
Use absolute paths for supplied input, configuration, output-root, and
reference files. When a stable prompt session identifier or references are
needed, pass the supported CLI flags. Supply credentials through Notarius's
documented configuration and environment mechanisms, never as command-line
arguments or generated secret-bearing configuration.
reference files. Notarius generates a stable prompt session for the resolved
input module and exact input bytes. Pass **--session-id** only when intentionally
grouping different invocations under a different session. Supply credentials
through Notarius's documented configuration and environment mechanisms, never
as command-line arguments or generated secret-bearing configuration. In
particular, a session identifier is provider-visible and is not a credential
mechanism.
Wait for the process before interpreting standard output. Only an exit status
of 0 permits decoding the receipt. On a nonzero exit, retain standard error for

View File

@@ -64,6 +64,8 @@ kind, and complete valid evidence. It does not infer turns, initiative, or
actions from registry or scene data.
The [NPC-interaction artifact](dnd-npc-interaction-artifacts.md) records
broader NPC occurrences. The [JSON output contract](json-output.md) defines
publication, and [D&D module internals](../internal/dnd.md) describes routing
and validation mechanics.
broader NPC occurrences. The [enemy-event artifact](dnd-enemy-event-artifacts.md)
uses combat turns as grounding only; turns do not establish an enemy event or
its outcome. The [JSON output contract](json-output.md) defines publication,
and [D&D module internals](../internal/dnd.md) describes routing and validation
mechanics.

View File

@@ -0,0 +1,114 @@
# D&D Enemy-Event Artifact
This contract defines the durable, source-grounded enemy-event occurrence list.
It records enemies directly established as opposing the party and explicitly
observed combat outcomes. It is an ordered observation artifact from which a
consumer may derive a ledger; it is not a ledger, encounter roster, or terminal
state model.
## Identity and compatibility
| Property | Value |
| --- | --- |
| Artifact kind | `dnd/enemy-event-list` |
| Schema ID | `notarius.dnd.enemy_events` |
| Schema name | `notarius_dnd_enemy_events_v1` |
| Schema version | `v1` |
| Media type | `application/json` |
`v1` is a strict JSON object with required `events`; the array may be empty.
Event and source-reference objects reject unknown fields. An incompatible shape
change requires a new schema version.
## Wire shape
Every event has these required fields:
| Field | Contract |
| --- | --- |
| `name` | Non-empty display name or directly grounded collective subject label. |
| `kind` | `engaged`, `killed`, `fled`, `captured`, or `incapacitated`. |
| `source_refs` | One or more current-transcript evidence ranges. |
Each source reference has exactly `source_id`, `start_unit_id`, and
`end_unit_id`. It identifies an inclusive current-transcript range; unit IDs
are positive and the start may not follow the end.
```json
{
"events": [
{
"name": "Ashfang",
"kind": "engaged",
"source_refs": [
{"source_id": "session-7", "start_unit_id": 41, "end_unit_id": 42}
]
},
{
"name": "Ashfang",
"kind": "fled",
"source_refs": [
{"source_id": "session-7", "start_unit_id": 57, "end_unit_id": 58}
]
}
]
}
```
## Event semantics and evidence
| Kind | Required evidence |
| --- | --- |
| `engaged` | The subject is directly established as actively opposing the party in combat. At most one engagement is emitted for one subject in one combat scene. |
| `killed` | The transcript explicitly establishes that the subject died or was killed. Damage, defeat, disappearance, or combat ending is insufficient. |
| `fled` | The subject explicitly escapes, retreats, or otherwise leaves combat to avoid continued engagement. Movement or absence from later turns is insufficient. |
| `captured` | The subject is explicitly taken prisoner or secured under the party's control. A grapple or temporary restraint alone is insufficient. |
| `incapacitated` | The subject is explicitly rendered unable to continue acting without being established as killed or captured. A missed turn is insufficient. |
The current transcript is the only event evidence. Campaign context and
normalized NPC, scene-description, combat-turn, and NPC-interaction artifacts
can ground names or control combat eligibility, but none may supply event
evidence. An outcome may share evidence with an engagement, in which case both
events are retained.
Extraction is limited to chunks with an exact combat-scene classification. An
exact non-combat classification produces an accepted empty list. Missing or
mismatched classification also produces an accepted empty list and a
`scene_classification_unavailable` warning.
## Subjects, normalization, and order
A subject matching the normalized NPC registry uses that registry's canonical
display name. Unmatched hostile creatures, summoned entities, and directly
grounded groups remain valid subjects. An unnamed homogeneous group uses the
narrowest transcript-grounded label, such as `Orcs`, `One orc`, or `Remaining
orcs`; the artifact never invents synthetic member identities or quantities.
Party members, allies, neutral observers, mentioned-but-absent enemies, hazards,
traps, and environmental effects are excluded.
Normalization collapses surrounding and repeated internal whitespace in subject
display values, canonicalizes recognized registry names, canonicalizes and
deduplicates exact source ranges, then orders events by valid evidence
chronology, normalized subject identity, display name, kind, and reference
sequence. The deterministic kind tie order is `engaged`,
`incapacitated`, `captured`, `fled`, then `killed`. Only entries with the same
normalized name, kind, and complete canonical evidence sequence are collapsed.
Different kinds, evidence, repeated engagement in separate scenes, and later
outcomes remain separate. A later engagement for the same named subject is
preserved after an earlier outcome because the artifact does not assert an
irreversible state transition.
## Non-goals
The artifact has no NPC or scene ID, quantity, confidence, description,
rationale, summary, current state, or inferred terminal outcome. It does not
emit `active` or `unresolved`; consumers may derive an unresolved ledger view
only when an engagement has no later explicit outcome. It never infers an
outcome from turn absence, scene termination, initiative order, hit-point
guesses, or other artifacts.
The [JSON output contract](json-output.md) defines publication. Configuration
keys, required generated-reference slots, and validator-chain selection are
defined in the [configuration reference](../config.md). Implementation and
prompt-grounding mechanics are described in the
[D&D module internals](../internal/dnd.md).

View File

@@ -0,0 +1,89 @@
# D&D Location Artifact
This contract defines the durable, source-grounded location registry produced
by `dnd/locations`. It records transcript-established physical places for one
source document; it is not a map, location hierarchy, campaign-wide world
registry, or location description.
## Identity and compatibility
| Property | Value |
| --- | --- |
| Artifact kind | `dnd/location-list` |
| Schema ID | `notarius.dnd.locations` |
| Schema name | `notarius_dnd_locations_v1` |
| Schema version | `v1` |
| Media type | `application/json` |
| Identity policy | `dnd.locations.identity.v1` |
`v1` accepts one strict JSON object with required `locations`; the array may be
empty. Location and source-reference objects reject unknown fields. An
incompatible artifact shape or identity-policy change uses a new version or
policy.
## Wire shape and identity
Each location has these required fields:
| Field | Contract |
| --- | --- |
| `id` | `location:sha256:` followed by 64 lowercase hexadecimal characters. |
| `name` | Non-empty transcript-established display name. |
| `source_refs` | One or more transcript evidence ranges that identify the place. |
A source reference has exactly `source_id`, `start_unit_id`, and `end_unit_id`.
The source ID identifies the transcript, unit IDs are positive inclusive unit
identifiers, and the start may not follow the end.
```json
{
"locations": [
{
"id": "location:sha256:5c1a91f15729df0b8c257093865fdf2452b43c215375e8cf2341aa9c37bb99aa",
"name": "Moon Gate",
"source_refs": [
{"source_id": "session-7", "start_unit_id": 4, "end_unit_id": 5}
]
}
]
}
```
The ID is deterministic and scoped to the source document. Notarius normalizes
the display name for comparison with Unicode NFKC, supported apostrophe
normalization, collapsed whitespace, and case folding. It hashes compact JSON
for this array, using the earliest canonical source reference as the anchor:
```text
["dnd.locations.identity.v1", comparison_name, source_id, start_unit_id, end_unit_id]
```
The canonical ID is the lowercase SHA-256 digest of those bytes with the
`location:sha256:` prefix. Equal display names are allowed when their evidence
anchors differ, so a generic name does not force distinct places to collapse.
## Scope, reconciliation, and evidence
Locations are physical or spatial places established by the transcript, such
as planes, regions, settlements, districts, buildings, rooms, landmarks,
routes, and geographic features. A generic label is permitted only when it
identifies a specific place in the transcript. Notarius does not infer an
unstated place or add hierarchy, coordinates, descriptions, participants, or
ownership.
Normalization first applies deterministic display, evidence, and ID rules. It
then may use a bounded LLM-assisted proposal to reconcile semantically duplicate
records. The proposal is validated and applied conservatively; invalid or
unusable proposals retain the deterministic result with retry or fallback
diagnostics. The registry's source references establish registry provenance,
not evidence for later artifacts.
## Consumers and publication
`dnd/location-occurrences` requires one approved location registry through its
`locations` reference slot. Its prompt receives an ordered source-free `{id,
name}` projection and must not treat registry references as occurrence
evidence. See the [location-occurrence artifact](dnd-location-occurrence-artifacts.md)
for that contract, [Configuration](../config.md#references-and-ordered-handoffs)
for binding rules, and the [JSON output contract](json-output.md) for
publication.

View File

@@ -0,0 +1,84 @@
# D&D Location-Occurrence Artifact
This contract defines the durable occurrence list produced by
`dnd/location-occurrences`. It records source-grounded ways the party relates
to locations in a required normalized location registry; it does not extend
that registry or infer a place absent from it.
## Identity and compatibility
| Property | Value |
| --- | --- |
| Artifact kind | `dnd/location-occurrence-list` |
| Schema ID | `notarius.dnd.location_occurrences` |
| Schema name | `notarius_dnd_location_occurrences_v1` |
| Schema version | `v1` |
| Media type | `application/json` |
`v1` accepts one strict JSON object with required `occurrences`; the array may
be empty. Occurrence and source-reference objects reject unknown fields. An
incompatible shape change requires a new schema version.
## Wire shape
Each occurrence has these required fields:
| Field | Contract |
| --- | --- |
| `location_id` | Exact ID from the required normalized [location registry](dnd-location-artifacts.md). |
| `name` | Exact canonical display name for `location_id` in that registry. |
| `kind` | One of `visited`, `planned`, `recalled`, or `mentioned`. |
| `source_refs` | One or more current-transcript evidence ranges for this occurrence. |
A source reference has exactly `source_id`, `start_unit_id`, and `end_unit_id`.
It identifies an inclusive range in the current transcript; unit IDs are
positive and the start may not follow the end.
```json
{
"occurrences": [
{
"location_id": "location:sha256:5c1a91f15729df0b8c257093865fdf2452b43c215375e8cf2341aa9c37bb99aa",
"name": "Moon Gate",
"kind": "visited",
"source_refs": [
{"source_id": "session-7", "start_unit_id": 12, "end_unit_id": 13}
]
}
]
}
```
## Occurrence categories
| Kind | Meaning |
| --- | --- |
| `visited` | The transcript establishes physical party presence, including arrival, continuing presence, or departure. |
| `planned` | The party explicitly proposes, intends, or agrees to future travel; speculation alone is not enough. |
| `recalled` | The transcript explicitly recounts prior party presence before the current live events. |
| `mentioned` | The location is explicit but no stronger category applies, including lore, directions, third-party activity, non-actionable speculation, a mere hypothetical reference, or out-of-character discussion. |
For overlapping evidence, precedence is `visited`, then `planned`, then
`recalled`, then `mentioned`. For example, “What if we went to Moon Gate?” is
eligible as `mentioned` when its narrow evidence explicitly references that
registry location, but it is not `planned` without an actual proposal,
intention, or agreement to travel. Inferred, unstated, uncertain, and
unsupported places or occurrences are omitted. Normalization
canonicalizes the registry name, orders and deduplicates source references, and
orders occurrences by source chronology, location ID, name, kind, and reference
sequence. It collapses only exact duplicates with the same ID, kind, and
complete canonical evidence sequence.
## Required grounding and evidence
Both extraction and normalization require exactly one `locations` reference of
kind `dnd/location-list`, media type `application/json`, and at most 1 MiB. The
registry provides identity grounding only: unknown IDs and mismatched ID/name
pairs are rejected rather than guessed or reassigned. The current transcript is
the only evidence source for an occurrence; registry evidence and provenance
never become occurrence evidence.
See [Configuration](../config.md#d-d-reference-slots) for the selectable slot
and generated-handoff compatibility, [D&D module internals](../internal/dnd.md)
for implementation behavior, and the [JSON output contract](json-output.md)
for publication.

View File

@@ -55,15 +55,26 @@ with the same canonical identity, retains their earliest position, and merges
their canonicalized evidence; it does not add aliases, roles, descriptions, or
relationship fields.
When evidence supports a semantically duplicate group, the canonical display
name is one of that group's supplied candidates. A complete, stable proper name
is preferred over an abbreviation. An unadorned proper name is preferred over
the same name plus a contextual class, role, title, or relationship descriptor
unless the transcript establishes that descriptor as part of the person's
name. A longer candidate is not preferred solely because it includes such a
descriptor.
## Scope and consumers
Only individually identifiable NPC names with transcript evidence belong in
this artifact. Groups, generic roles, invented labels, and descriptive
enrichment are excluded. Its source references prove registry provenance; they
do not become evidence for a spell, interaction, or combat occurrence.
do not become evidence for a spell, interaction, combat, or enemy-event
occurrence.
This registry can ground actor or caster names in the [spell](dnd-spell-artifacts.md)
and [combat-turn](dnd-combat-turn-artifacts.md) artifacts. It is required to
resolve the canonical `name` in an [NPC interaction](dnd-npc-interaction-artifacts.md).
The [enemy-event artifact](dnd-enemy-event-artifacts.md) also uses it only for
subject grounding and canonical display names.
The [JSON output contract](json-output.md) defines publication, and
[D&D module internals](../internal/dnd.md) owns pipeline mechanics.

View File

@@ -74,5 +74,8 @@ Only entries with the same canonical name, kind, and complete valid evidence
sequence are collapsed; distinct categories or evidence remain separate.
See the [combat-turn artifact](dnd-combat-turn-artifacts.md) for combat-action
occurrences and the [JSON output contract](json-output.md) for publication.
Pipeline mechanics are described in [D&D module internals](../internal/dnd.md).
occurrences. The [enemy-event artifact](dnd-enemy-event-artifacts.md) consumes
only `combat_opponent` interactions as grounding; they never establish an enemy
event or outcome. The [JSON output contract](json-output.md) defines
publication. Pipeline mechanics are described in
[D&D module internals](../internal/dnd.md).

View File

@@ -62,8 +62,9 @@ durable fields, or the same source range with different kind, title, or
summary, is invalid. It does not merge adjacent ranges, alter prose, or infer
missing scenes.
The [combat-turn artifact](dnd-combat-turn-artifacts.md) uses an exact matching
The [combat-turn artifact](dnd-combat-turn-artifacts.md) and
[enemy-event artifact](dnd-enemy-event-artifacts.md) use an exact matching
`combat` scene only as eligibility control; scene title, summary, and source
reference never become combat evidence. Publication is defined by the
reference never become their evidence. Publication is defined by the
[JSON output contract](json-output.md); implementation details live in
[D&D module internals](../internal/dnd.md).

View File

@@ -74,8 +74,11 @@ than infer a lane schema from its name. The current D&D payload contracts are
[spells](dnd-spell-artifacts.md), [NPCs](dnd-npc-artifacts.md),
[NPC interactions](dnd-npc-interaction-artifacts.md),
[combat turns](dnd-combat-turn-artifacts.md),
[item events](dnd-item-event-artifacts.md), and
[scene descriptions](dnd-scene-description-artifacts.md).
[item events](dnd-item-event-artifacts.md),
[scene descriptions](dnd-scene-description-artifacts.md), and
[enemy events](dnd-enemy-event-artifacts.md),
[locations](dnd-location-artifacts.md), and
[location occurrences](dnd-location-occurrence-artifacts.md).
## `manifest.json`
@@ -98,6 +101,11 @@ summarize results without embedding lane payload bytes. A chunk-plan summary is
provenance for the plan used by this run; cache records, debug artifacts, and
other operational state are not published as bundle files.
When present, `metadata.session_id` is the effective non-secret routing
correlation identifier used for the run. It can be visible to providers and is
not a substitute for a cache or checkpoint identity. Its generation and
override behavior are defined by the [CLI reference](../cli.md#run).
Each `llm_profiles` entry identifies effective, non-secret LLM execution
provenance:

View File

@@ -48,11 +48,13 @@ adapter boundary. It also retains responsibility for pipeline retries,
scheduling, debug persistence, redaction, profile provenance, and conversion
from private model responses into durable domain artifacts.
Notarius sends its trimmed run session through PromptKit's direct session
Notarius sends one stable effective session through PromptKit's direct session
field, which is authoritative for provider session behavior. It also retains
the same value as the `session_id` prompt variable for maintained prompt
compatibility. Session IDs are stable, non-secret correlation identifiers and
may be exposed to providers and provider observability.
compatibility. The generated identifier is 76 ASCII characters, within
PromptKit v0.5.0's 256-code-point session limit. Session IDs are non-secret
correlation identifiers and may be exposed to providers and provider
observability. The CLI contract owns generation and override behavior.
Notarius records PromptKit's selected backend ID and effective reasoning
setting as optional run-manifest provenance. Endpoint-only profiles have no

View File

@@ -89,9 +89,11 @@ handoff:
profiles;
4. materialize external or generated references and record redacted invocation
and resolution provenance when debug capture is enabled;
5. construct registries, the scheduled LLM client, prepared modules, and the
requested cache/checkpoint collaborators;
6. read the source input and invoke the framework runner; and
5. construct registries, the scheduled LLM client, and prepared modules;
6. read the source input once, resolve its effective session from the explicit
override or resolved input module and raw bytes, then construct requested
checkpoint collaborators and invoke the framework runner with that same
value; and
7. write the runner's logical output files only after a successful run, then
complete the command report and user-facing result.
@@ -102,6 +104,13 @@ final command result. Detailed state lifecycle, resume handling, and physical
path confinement are maintained in [Run State Internals](state.md) and
[Operations](../operations.md).
The CLI owns the versioned generated-session policy and resolves the sole
effective value before checkpoint construction. It records that value in the
final debug invocation summary when capture is enabled and passes it unchanged
to checkpoint identity and `pipeline.RunInput`. The public flag and stability
contract are defined by the [CLI reference](../cli.md#run); framework and LLM
packages only transport the supplied value.
For `run --json`, the CLI constructs and encodes its private run-result receipt
after a successful runner result is available, before it publishes logical
output files. It writes the prepared receipt to standard output only after

View File

@@ -7,7 +7,7 @@ selectable keys, bindings, reference syntax, and default validator chains.
## Durable Artifact Contracts
The six lanes have separate durable wire contracts. This guide deliberately
The nine lanes have separate durable wire contracts. This guide deliberately
does not repeat their JSON shapes or schemas.
| Lane | Durable contract |
@@ -18,6 +18,9 @@ does not repeat their JSON shapes or schemas.
| Item events | [item-event artifacts](../integrations/dnd-item-event-artifacts.md) |
| NPC interactions | [NPC-interaction artifacts](../integrations/dnd-npc-interaction-artifacts.md) |
| Scene descriptions | [scene-description artifacts](../integrations/dnd-scene-description-artifacts.md) |
| Enemy events | [enemy-event artifacts](../integrations/dnd-enemy-event-artifacts.md) |
| Locations | [location artifacts](../integrations/dnd-location-artifacts.md) |
| Location occurrences | [location-occurrence artifacts](../integrations/dnd-location-occurrence-artifacts.md) |
## Family Composition
@@ -25,9 +28,9 @@ The D&D registrar registers the familys artifact codecs, extractors, typed
append-order mergers, normalizers, validators, prompt assets, fallback LLM
profile asset, and default validator chains. Each extractor and normalizer has
a stable module spec, explicit execution class, strict option decoding, and a
typed builder. Scene chunking, every extractor, and NPC normalization are
registered as `llm_backed`; the remaining current D&D mergers and normalizers
are `deterministic`. The metadata is available to catalog inspection and
typed builder. Scene chunking, every extractor, NPC normalization, and location
normalization are registered as `llm_backed`; the remaining current D&D mergers
and normalizers are `deterministic`. The metadata is available to catalog inspection and
resolved-pipeline debug data and determines which selected bindings inherit the
pipeline profile. Configuration remains the canonical owner of the exact keys,
profile precedence, and validator order.
@@ -40,14 +43,21 @@ the contracts above define durable data.
## Prompt Construction
D&D LLM-facing content lives beneath `assets/dnd/`. New extractor content uses
its feature subtree; when a family has both extraction and normalization
content, keep those in its `extract` and `normalize` subtrees. The owning module
still defines the ordered manifest and registers the resulting scoped filesystem.
Shared fragments belong to the D&D shared implementation and are selected by
name, never copied into individual module subtrees.
D&D extractors assemble prompts from an ordered manifest of shared and
module-owned assets. Reuse the shared D&D system, evidence, identity,
reference, and transcript assets instead of copying their text into individual
modules. A manifests declared sequence, including cache-control placement, is
part of the prompt behavior.
module-selected assets. The location extractor and occurrence extractor reuse
the shared D&D system, evidence, identity, reference, and transcript assets
instead of copying their text into individual modules. A manifests declared
sequence, including cache-control placement, is part of the prompt behavior.
Every maintained D&D LLM prompt selects `dnd-extraction` as its default
profile. The D&D registrar embeds that fallback profile with the maintained
profile. The D&D registrar registers that fallback profile with the maintained
OpenRouter model, timeout, and service-tier policy. An operator may provide a
complete profile with the same ID through the configured PromptKit source; that
definition replaces the fallback rather than merging with it. The fallback
@@ -71,7 +81,9 @@ remains stable.
The other D&D LLM prompts intentionally follow different patterns. Scene
chunking has no sibling extraction lane with which to share its full transcript,
so it renders campaign references before its task and instructions, then places
the cacheable full transcript last. NPC normalization keeps its task and
the cacheable full transcript last. NPC and location normalization share the
entity-reconciliation response schema and safety boundary while retaining their
own task and identity rules. NPC normalization keeps its task and
cacheable instructions before the candidate collection, followed by the
cacheable transcript windows: candidates must be available before their
supporting evidence is evaluated, and those windows are not a cross-lane
@@ -101,14 +113,20 @@ relatedness validators report advisory evidence concerns. The configured order
is documented in
[Configuration](../config.md#production-validator-keys-and-default-chains).
Enemy-event extraction additionally rejects a second `engaged` observation for
the same comparison identity within one scene-scoped result. Normalization may
combine results from distinct scenes, so it intentionally does not apply that
rule. Configuration owns the exact validator key and chain position.
Normalizers are deterministic for spells, combat turns, item events, NPC
interactions, and scene descriptions. They canonicalize display values and
evidence, use source-document order for stable output, and issue bounded
warnings for changes or collapsed duplicates. The NPC normalizer is the
intentional exception: it first produces a deterministic candidate set, then
uses a bounded structured-LLM proposal to reconcile identity groups. Invalid
or unusable proposals retain the deterministic result and surface retry or
fallback diagnostics; the model does not directly replace durable records.
interactions, scene descriptions, enemy events, and location occurrences. They canonicalize display
values and evidence, use source-document order for stable output, and issue
bounded warnings for changes or collapsed duplicates. The NPC and location
normalizers are intentional exceptions: each first produces a deterministic
candidate set, then may use a bounded structured-LLM proposal to reconcile
identity groups. Invalid or unusable proposals retain the deterministic result
and surface retry or fallback diagnostics; the model does not directly replace
durable records.
## Generated References And Grounding
@@ -122,7 +140,13 @@ NPC registries are names-only grounding projections: they may canonicalize
actors for spells and combat turns and are required for NPC interactions, but
they do not supply evidence. Scene-description registries are eligibility-only
projections: they retain the current chunks classification data, not scene
prose or evidence, and exist to route combat extraction.
prose or evidence, and exist to route combat extraction. Enemy-event extraction
also projects combat turns to `actor` and `turn_kind` and filters NPC
interactions to `combat_opponent` names and kinds. Location registries project
ordered `{id, name}` pairs to location-occurrence extraction and normalization;
exact ID/name matching keeps same-name locations distinguishable. These compact
projections, like NPC grounding, are source-free guidance and never event
evidence.
## Lane-Specific Rules
@@ -137,11 +161,17 @@ shared helper changes.
| Item events | Uses campaign context for disambiguation but has no NPC-registry or scene-description dependency. |
| NPC interactions | Requires the normalized NPC registry at extraction and normalization, using it for canonical actor grounding only. |
| Scene descriptions | Produces the classifications consumed by combat routing; it does not consume an NPC registry or provide evidence for combat artifacts. |
| Enemy events | Requires NPC, scene-description, combat-turn, and NPC-interaction artifacts. It calls the LLM only for an exact `combat` classification, records ordered observations rather than terminal state, and normalizes recognized names through the NPC registry while preserving grounded collective labels. |
| Locations | Produces a source-anchored, session-scoped registry. Its LLM-assisted reconciliation is proposal-only and never collapses same-name places without validated identity and evidence rules. |
| Location occurrences | Requires the normalized location registry for both extraction and normalization. Its [durable occurrence categories](../integrations/dnd-location-occurrence-artifacts.md#occurrence-categories) distinguish explicit speculation from unsupported inference; the deterministic normalizer enforces exact registry grounding and never turns registry provenance into occurrence evidence. |
The combat and scene-description contracts describe their exact handoff and
empty-result behavior in more detail:
[combat turns](../integrations/dnd-combat-turn-artifacts.md) and
[scene descriptions](../integrations/dnd-scene-description-artifacts.md).
The [enemy-event contract](../integrations/dnd-enemy-event-artifacts.md)
defines its durable semantics; [Configuration](../config.md) owns its
selectable bindings and validation chains.
## Focused Verification

View File

@@ -26,9 +26,10 @@ durable schemas. Those responsibilities remain with the module and its
`PromptKitClient` validates the request target and prompt identity, maps each
named material to a PromptKit inline artifact while preserving its origin URI,
maps the trimmed request session to PromptKit's direct per-run session field,
retains the same value as the `session_id` prompt variable for maintained
prompt compatibility, and forwards profile selection. It then creates one
passes the supplied request session through to PromptKit's direct per-run
session field, retains the same value as the `session_id` prompt variable for
maintained prompt compatibility, and forwards profile selection. It does not
derive or replace session values; the CLI owns that policy. It then creates one
frozen prepared execution, captures its caller-owned credential-redacted
details for debug material, and executes that exact snapshot through
PromptKit's prepared-execution boundary. The direct field
@@ -124,16 +125,24 @@ the corresponding PromptKit filesystems and rejects invalid roots, unreadable
assets, duplicate paths, and missing prompt or schema files during preparation.
Fallback assets receive a safe content digest for checkpoint identity; raw
paths and bytes are never included. The frameworks `promptfs` helper combines
module-owned prompt files with reusable domain fragments without making the
module-selected prompt files with reusable domain fragments without making the
framework depend on D&D content.
Each LLM-backed module owns its prompt declaration, package-specific assets,
and private response schema. Shared D&D wording is owned by the D&D shared
asset package; the detailed D&D conventions are in
[D&D Module Internals](dnd.md). The mounted prompt assets used by a module also
determine its prompt fingerprint. Schema loaders validate JSON, attach identity
and digest metadata, make defensive copies, and expose diagnostics without raw
schema bytes.
LLM-facing content is embedded once by the root `assets` package. Each consumer
uses only its scoped subtree, while the module retains ownership of its prompt
declaration, ordered manifest, private response-schema identity, and
registration. Shared D&D fragments are selected by D&D's shared implementation;
the detailed convention is in [D&D Module Internals](dnd.md). This physical
arrangement and its data-only boundary are defined by
[Architecture](../policy/architecture.md) and
[ADR-0011](../adr/0011-centralize-llm-assets.md), rather than by this runtime
guide.
Mounted prompt assets determine a module's fingerprint. The fingerprint hashes
only the module and shared files explicitly selected by its manifest, so an
unrelated asset does not invalidate a checkpoint. Schema loaders validate JSON,
attach identity and digest metadata, make defensive copies, and expose
diagnostics without raw schema bytes.
Private response schemas validate a model transport envelope. They are not the
durable artifact schema and should not be documented as an external wire

View File

@@ -29,6 +29,7 @@ physical state roots.
| Generic models | **internal/core/source**, **internal/core/artifacts**, **internal/framework/contracts** | Source documents and chunks, manifests and provenance, plus typed artifact, reference, validation, output, and structured-completion contracts. |
| Pipeline framework | **internal/framework/pipeline** | Registries, profile and reference resolution, typed preparation, validation, retry coordination, ordered execution, handoff, and result assembly. |
| LLM and prompt runtime | **internal/framework/llm**, **internal/framework/promptfs** | Provider-neutral structured completions, scheduling, profile recording, prompt assets, schema registration, and credential-shaped-value redaction. |
| Embedded LLM content | **assets** | Read-only centralized LLM-facing content, scoped by its consuming package; see [LLM Runtime](llm.md#prompt-and-schema-assets) and [D&D Module Internals](dnd.md#prompt-construction). |
| Runtime state | **internal/core/fileio**, **internal/core/debugbundle**, **internal/framework/checkpoint**, **internal/framework/chunkplan**, **internal/framework/chunkmap**, **internal/framework/debug** | Confined atomic files, debug bundles, checkpoint and chunk-plan state, accepted chunk maps, and pipeline-facing debug recording. |
| Production extensions | **internal/modules/generic**, **internal/modules/seriatim**, **internal/modules/dnd** | Domain-neutral extensions, Seriatim input support, and D&D extraction families registered into the production catalog. |

View File

@@ -10,11 +10,11 @@ own durable output shapes. Concrete production extensions are covered by
## Boundary
The pipeline framework accepts a resolved composition, registries, shared
dependencies, input bytes, and state/debug collaborators. It returns logical
output files, normalized artifacts, recorded rejections and warnings, manifest
provenance, and checkpoint decisions. The CLI owns process arguments,
configuration discovery, physical roots, and placement of returned output
files.
dependencies, input bytes, a supplied prompt session, and state/debug
collaborators. It returns logical output files, normalized artifacts, recorded
rejections and warnings, manifest provenance, and checkpoint decisions. The
CLI owns process arguments, configuration discovery, session resolution,
physical roots, and placement of returned output files.
The framework has one fixed shape:
@@ -84,6 +84,10 @@ incompatible producer prevents the consumer step from starting.
The runner validates its input, installs no-op state collaborators when none
were supplied, and serially performs source parsing and chunk-plan selection.
It transports the supplied session unchanged to prompt-facing operations and
run-manifest metadata; it neither derives a session nor substitutes a parsed
source document identifier. The public session contract is owned by the
[CLI reference](../cli.md#run).
An accepted plan is materialized into source-addressed chunks and passes the
configured chunk validators before any lane runs. A chunk rejection is a
recorded pipeline outcome: lanes do not start, but the output stage can encode

View File

@@ -266,16 +266,18 @@ transport-wide cap. Notarius does not add another timeout around PromptKit.
The pinned upstream boundary and profile-format links are in
[PromptKit Integration](integrations/pkg-promptkit.md).
Concurrency has two independent layers. Notarius **total_llm** is the
application-wide provider-call limit shared by all backends, modules, retries,
and validators. PromptKit may impose a narrower admission limit for the
selected backend. The effective active-generation bound is the intersection of
both limits and can therefore be lower than **total_llm**. Built-in OpenRouter
profiles use PromptKit's upstream backend limit; endpoint-only profiles have no
PromptKit backend limit and remain bounded by Notarius. For the configured
local backend, a zero **concurrency_limit** leaves only the Notarius scheduler
as a call limit. A positive value makes the effective active local-generation
bound the smaller of **total_llm** and that local limit.
Concurrency has two independent layers. Notarius **total_llm** defaults to 16
and is the application-wide provider-call limit shared by all backends,
modules, retries, and validators. PromptKit may impose a narrower admission
limit for the selected backend. The effective active-generation bound is the
intersection of the Notarius limit, any PromptKit backend limit, and work made
available by the pipeline. Built-in OpenRouter profiles use PromptKit's
upstream backend limit; endpoint-only profiles have no PromptKit backend limit
and remain bounded by Notarius. For the configured local backend, a zero
**concurrency_limit** leaves only the Notarius scheduler as a call limit. A
positive value makes the effective active local-generation bound the smaller
of **total_llm** and that local limit, so a local limit of four permits no more
than four active local generations.
For a positive local limit, PromptKit owns its default waiting capacity and
admission behavior. When a PromptKit backend has admitted all active and queued
@@ -289,3 +291,11 @@ under [PromptKit profiles](config.md#promptkit-profiles) and
limits and actual provider-call limits are independent. Notarius writes local
filesystem state only; remote storage, archival, and retention automation are
outside the implemented CLI.
Every run has an effective prompt session used for provider routing and run
provenance. The generated default is stable for the same input module and raw
input bytes; use [**--session-id**](cli.md#run) only when intentionally grouping
different invocations. Both generated and explicit values can be visible to
providers, manifests, checkpoints, and requested debug bundles. Do not put
credentials or other secrets in an explicit session identifier; command-line
values are not a credential mechanism.

View File

@@ -44,6 +44,13 @@ must not compose the application or take ownership of process behavior. The
current packages implementing these layers are inventoried in
[Internal Overview](../internal/overview.md).
The root `assets` package is a content-only dependency leaf. It may expose a
read-only embedded filesystem, but it must contain no business logic and must
not depend on `internal` packages or PromptKit. Consumers scope that filesystem
to the content they own; the root package is not a behavioral registry or a
public extension contract. The rationale and compatibility consequence are
recorded in [ADR-0011](../adr/0011-centralize-llm-assets.md).
The following dependency boundaries are mandatory:
- extractors and validators do not depend on concrete input adapters;
@@ -73,6 +80,12 @@ Extract modules own artifact semantics, prompt use, response schemas, and
domain interpretation. Domain-specific concepts remain in the relevant module,
validator, shared domain helper, and artifact contract.
Physical centralization of LLM-facing content does not transfer semantic
ownership from those modules. Modules retain their manifests, response-schema
identities, prompt ordering, and registration, while reading only their scoped
content subtree. Generic framework code remains domain-neutral when it reads
its own scoped generic assets from the shared content container.
Typed artifact registrations declare one stable artifact kind and exact Go
type from extraction through merge, normalization, and semantic validation.
Pipeline resolution requires a compatible codec and matching kind-specific

View File

@@ -7,37 +7,6 @@ not as committed release dates.
## Near-Term D&D Pipeline
### Combat Enemy Ledger
- Add a D&D artifact that identifies enemies faced during combat and supports
an end-of-session encounter ledger.
- Track each enemy's observed state using a small controlled vocabulary such as
`active`, `killed`, `fled`, `captured`, or `incapacitated`, while preserving
an explicit unresolved state when the transcript does not establish an
outcome.
- Preserve the evidence for enemy participation and state changes rather than
inferring a terminal outcome from combat ending or an enemy disappearing
from the conversation.
- Define how repeated mentions, groups of unnamed enemies, summoned or allied
creatures, and the same enemy appearing in multiple combats affect identity
and ledger entries.
- Evaluate whether the ledger should be extracted directly, derived from
combat-turn artifacts, or use a sequential pipeline that consumes combat
turns and the normalized NPC registry as grounding references.
### Location Extraction
- Add a D&D artifact for locations visited by the party or otherwise mentioned
in the transcript.
- Distinguish observed visits from references, plans, recalled places, and
uncertain or inferred locations so a mention alone is not reported as a
visit.
- Preserve transcript evidence for each visit or mention and reconcile aliases,
nested places, and repeated appearances without collapsing distinct
locations that share a generic name.
- Define how the location artifact should ground later narrative reports and
whether future event artifacts should retain canonical location identities.
### Evaluate Spell Extraction And Normalization
- Evaluate ordinary extraction retries and the completed normalization path
@@ -53,71 +22,6 @@ not as committed release dates.
spell, combat, interaction, and scene-description lanes after real-world use.
Add more complex chunking only in response to demonstrated failures.
## Cross-Cutting LLM Runtime
### Deterministic Prompt Session Identity
- Replace the source-document-ID default for prompt sessions with one
predictable, procedurally generated session ID for the complete
source-processing workload.
- Preserve an explicit non-empty `--session-id` as the highest-precedence
override. Otherwise, derive the default only from the effective input module
identity and the exact raw input bytes.
- Use a versioned, bounded representation such as
`notarius:v1:<sha256(input-module + NUL + raw-input)>`. The exact encoding
must fit PromptKit's session length contract and must not embed source
content.
- Keep the derived session stable across runs, pipelines, selected lanes,
ordered steps, retries, resume, recomputation, LLM profiles, reasoning
overrides, and output, debug, or cache settings.
- Do not include file-backed references, generated references, reference
contents, or the composition of a reference bundle in session derivation.
References may change between prompt calls within one pipeline without
changing routing affinity.
- Resolve the authoritative session before checkpoint construction and use the
same value for checkpoint runtime identity, every prompt-facing module,
PromptKit's direct session field, the compatibility `session_id` prompt
variable, run-manifest metadata, and debug metadata.
- Keep routing identity separate from cache and checkpoint content identity.
Exact prompt prefixes, reference contents, model settings, and other
generation-affecting inputs must continue to participate in their existing
hashes and checkpoint fingerprints even though they do not change the
session.
- Treat the generated value as a provider-visible, stable pseudonymous
correlation identifier. Do not introduce an installation-specific HMAC or
secret unless a concrete multi-tenant or privacy requirement justifies
sacrificing deterministic identity across installations.
### Raise The Default Application-Wide LLM Limit
- Raise the default `concurrency.total_llm` value from 1 to 16 so ordinary
single-backend runs can use PromptKit's expected OpenRouter capacity and
lower-capacity local backends without an unnecessarily narrower Notarius
limit.
- Keep the Notarius application-wide scheduler mandatory and require
`total_llm` to remain a positive integer. Do not make the default unlimited:
endpoint-only profiles, an unrestricted local backend, injected clients, and
aggregate work across several backends may have no narrower PromptKit limit.
- Continue defaulting `concurrency.stage_workers.extract` to the effective
`total_llm`, making its default 16 as part of the same change. Preserve an
explicit lower extract-worker setting when an operator wants less queued or
concurrent extraction work.
- Define effective provider concurrency as the intersection of the Notarius
application-wide limit, the selected PromptKit backend limit when present,
and the work made available by stage execution. A Notarius limit of 16 does
not narrow a backend already limited to 16, while a local backend limited to
4 remains bounded at 4.
- Treat the default as an application-wide safety ceiling across profiles,
backends, modules, retries, and validators. A run that intentionally needs
the combined capacity of several backends may configure a higher
`total_llm` and an appropriate extract-worker count explicitly.
- Retain the existing configuration and environment override surfaces. Update
canonical configuration, operations, and internal documentation together
when the default changes.
- Reconsider decoupling the extract-worker default from `total_llm` only after
mixed-backend workloads demonstrate a need for a high global emergency
ceiling with a lower default work-production rate.
## Shared Normalization And Quality Work
### Generic LLM-Assisted Deduplication

View File

@@ -1,586 +0,0 @@
# PromptKit v0.5 Implementation Plan
## Objective
Implement the target state in
[PromptKit v0.5 Integration And LLM Profile Policy](promptkit.md). Each numbered
stage is intended to be one implementation prompt for a GPT-5.6-Terra coding
agent. Complete stages in order and leave the repository buildable, tested, and
internally coherent after every stage.
Follow [Architecture](../policy/architecture.md),
[Testing Policy](../policy/testing.md), and
[Documentation Policy](../policy/documentation.md) throughout. Preserve
unrelated user changes. Use `apply_patch` for source and documentation edits,
run `gofmt` on changed Go files, and add only tests that protect the behaviors
and risks assigned to that stage.
Do not implement the separate deterministic session-ID or default-concurrency
roadmap items as part of this plan. Do not perform paid or credentialed LLM
calls.
## Background Summary
Notarius currently pins PromptKit v0.3.0, calls `Prepare` and then `Run` for one
completion, validates profiles through a synthetic prompt, has no application
fallback profile source, and accepts LLM profiles only at individual bindings
or through the run-wide CLI override. PromptKit v0.5.0 is source-compatible
with the current tree; a temporary v0.5.0 module override has already passed
`go test ./...`.
The implementation must nevertheless treat the upstream optional-parameter
change as intentional: unset `temperature`, `max_tokens`, and `top_p` remain
unset and are omitted from compatible provider requests. Do not restore the old
implicit `top_p: 1` default.
## Stage 1: Upgrade The PromptKit Dependency
### Goal
Establish a clean PromptKit v0.5.0 baseline before adopting its new APIs.
### Work
- Update `go.mod` and `go.sum` from PromptKit v0.3.0 to v0.5.0 and run
`go mod tidy`.
- Change the PromptKit built-in profile-catalog marker in
`internal/framework/llm/promptkit_profile_fingerprint.go` to identify
v0.5.0. This deliberately invalidates LLM checkpoints tied to the prior
catalog identity.
- Review PromptKit-facing compile errors or test failures against the v0.4.0
and v0.5.0 release guides. Do not adopt prepared execution, inspection, or
fallback profiles in this stage.
- Replace the existing test assertion for one exact built-in fingerprint hash
with durable assertions that the fingerprint is deterministic, non-empty,
non-secret, and changes when a semantic profile source changes. Do not add a
new version-constant or exact-hash change detector.
- Update `docs/integrations/pkg-promptkit.md` to pin and link v0.5.0 and state
the implemented dependency-level behavior: unset optional sampling controls
are provider defaults. Do not document later stages as implemented.
- Update any other canonical text that explicitly claims the dependency is
v0.3.0, but defer descriptions of unimplemented v0.5 APIs.
### Tests And Validation
- `go test ./internal/framework/llm ./internal/cli`
- `go test ./...`
- `go vet ./...`
- `go build ./cmd/notarius`
- `rg -n 'promptkit v0\.3\.0|promptkit@v0\.3\.0|PromptKit v0\.3\.0' .`
- `git diff --check`
### Completion Criteria
- The repository directly pins v0.5.0 and all default offline checks pass.
- The profile-source fingerprint identifies the new upstream catalog without a
brittle literal-hash test.
- Current documentation no longer identifies v0.3.0 as the supported version.
## Stage 2: Execute One Frozen Prepared Snapshot
### Goal
Make Notarius debug details and generation use one exact PromptKit preparation.
### Work
- Refactor `PromptKitClient.CompleteStructured` to call
`PrepareExecution`, immediately defer `Discard`, obtain a caller-owned
`Details` value, and execute with `RunPrepared`.
- Preserve the existing Notarius request mapping, cancellation precedence,
validation classification, raw structured bytes, response decoding,
profile recording, usage reporting, and credential redaction.
- Ensure every preparation, execution, validation, empty-result, and decode
error retains useful Notarius prompt context without exposing prepared handle
state or secrets.
- Use `errors.As` to obtain `*promptkit.CapacityError` on admission rejection.
Preserve `contracts.ErrLLMCapacityExceeded` as the stable classification and
add a nonblank backend ID only to safe application-owned diagnostic context.
Do not expose `promptkit.CapacityError` outside the LLM adapter.
- Update `docs/internal/llm.md` and the implemented-mechanics portion of
`docs/integrations/pkg-promptkit.md` to describe the single frozen execution
snapshot and structured capacity adaptation.
### Tests And Validation
- Adapt existing PromptKit client tests to the prepared-execution path.
- Retain or add one behavioral test proving that the debug prompt details match
the request actually passed to generation when a backing prompt source could
otherwise change between independent preparations. Test the resulting
snapshot consistency, not a private helper call count.
- Retain capacity tests proving `errors.Is` reaches
`contracts.ErrLLMCapacityExceeded`, the selected backend can appear in safe
diagnostic context, and provider calls are not made after rejected
admission.
- Run `go test ./internal/framework/llm` and
`go test -race ./internal/framework/llm`.
- Run `go test ./...` and `git diff --check`.
### Completion Criteria
- `CompleteStructured` no longer calls independent `Prepare` and `Run`
operations for one request.
- Debug prompt material and generation result originate from the same frozen
PromptKit snapshot.
- Capacity remains a provider-neutral Notarius error classification.
## Stage 3: Replace Synthetic Profile Validation With Inspection
### Goal
Validate profiles through PromptKit's exact profile-inspection boundary and
centralize engine profile-source construction.
### Work
- Introduce a small provider-adapter-owned profile inspection or validation
function in `internal/framework/llm`. Its public internal signature must use
Notarius-owned configuration and result/error types rather than returning
PromptKit types to the CLI.
- Share the code that applies `profile_dir`, `profile_file`, and registered
backend options between the production PromptKit engine and the inspection
engine. Preserve the mutual-exclusion and local-backend rules.
- Change CLI explicit-profile preflight to use `Engine.InspectProfile` through
that LLM boundary.
- Remove `profileCheckPromptID`, `profileCheckPromptFS`, the `testing/fstest`
production dependency, and the synthetic `Prepare` request.
- Preserve distinct, useful errors for an absent profile, invalid profile,
unknown backend registration, cancellation, and invalid profile source.
- Do not require `api_key_env` to be populated during configuration validation.
Inspection may report credential requirements internally, but actual
preparation remains responsible for credential availability before a model
call.
- Update current-behavior sections in `docs/internal/cli.md` and
`docs/internal/llm.md`. Keep field definitions in `docs/config.md`.
### Tests And Validation
- Replace synthetic-prompt tests with profile inspection tests covering:
configured local backend success; missing local backend failure; absent
profile; malformed profile; and an otherwise valid profile whose credential
environment variable is intentionally unset.
- Prove validation performs no provider HTTP call and remains offline.
- Run `go test ./internal/framework/llm ./internal/cli` and `go test ./...`.
- Run `git diff --check`.
### Completion Criteria
- No production synthetic profile-check prompt remains.
- Profile validation uses the same ordinary profile source and backend
registrations as execution.
- Configuration validation succeeds for structurally valid profiles without
reading credential values.
## Stage 4: Add Application Fallback Profile Asset Plumbing
### Goal
Allow module families to register application-owned fallback profile YAML
without placing domain policy in generic LLM code.
### Work
- Extend `internal/framework/llm.AssetRegistry` with a separate fallback
profile source collection, registration method, flattened filesystem, and
safe content digest.
- Reuse the existing asset-source path validation and flattening behavior where
appropriate. Reject invalid roots, unreadable assets, and duplicate flattened
paths. Do not parse PromptKit profile YAML in Notarius.
- Add `promptkit.WithFallbackProfileFS` to production engine options only when
at least one fallback profile source is registered.
- Supply the identical assembled fallback source to the profile-inspection
engine. Adjust CLI composition so pipeline-aware profile validation can use
the production LLM asset registry without exposing PromptKit types.
- Extend profile-source checkpoint identity to include the exact fallback
profile asset digest in addition to the PromptKit catalog marker and operator
source. Keep the resulting fingerprint hash-only and path/content/credential
free.
- Keep operator source precedence owned by PromptKit. Do not implement profile
merging or duplicate PromptKit source resolution in Notarius.
- Update `docs/internal/llm.md` only for the new implemented generic asset and
fingerprint mechanics. No domain fallback exists until Stage 5.
### Tests And Validation
- Add focused AssetRegistry tests for successful flattening, invalid roots,
duplicate paths, and hash changes when fallback bytes change.
- Add adapter-level tests showing that the fallback filesystem reaches both
execution construction and inspection construction.
- Extend checkpoint tests to prove fallback content changes profile-source
identity without exposing raw YAML or paths. Use relational comparisons, not
a fixed hash literal.
- Run `go test ./internal/framework/llm ./internal/cli` and `go test ./...`.
- Run `git diff --check`.
### Completion Criteria
- Generic plumbing can carry application fallback profiles while remaining
unaware of D&D IDs or model settings.
- Inspection, execution, and checkpoint identity use the same fallback asset
source.
## Stage 5: Adopt The D&D `dnd-extraction` Fallback
### Goal
Give the D&D module family one stable embedded workload profile that operators
can replace.
### Work
- Add a D&D-owned embedded PromptKit profile asset with ID `dnd-extraction`
under `internal/modules/dnd`. Use the exact baseline defined in
`promptkit.md`: OpenRouter, `openai/gpt-5.6-luna`, no explicit reasoning
effort, a 240-second timeout, flex service tier, and no selected temperature,
token limit, or `top_p`. The omitted reasoning value intentionally allows
OpenAI's backend to apply its `medium` default.
- Register the profile filesystem from the D&D registrar through the generic
fallback profile asset boundary. Keep D&D policy out of
`internal/framework/llm` and the CLI composition root.
- Change every maintained D&D LLM prompt definition—including scene chunking,
all D&D extractors, and NPC normalization—from the model-named default to
`default_profile: dnd-extraction`.
- Add an integration-level profile-resolution test proving that:
- the fallback resolves when no operator source defines the ID;
- a valid operator profile with the same ID wins completely; and
- an invalid matching operator profile fails rather than falling through.
- Test through Notarius's assembled production assets and PromptKit boundary;
do not duplicate every upstream source-precedence case.
- Update the implemented profile ownership and prompt-default behavior in
`docs/internal/dnd.md`, `docs/internal/llm.md`, and
`docs/integrations/pkg-promptkit.md`. Defer the complete operator walkthrough
and examples to Stage 10.
### Tests And Validation
- Run focused D&D prompt preparation tests and the production composition
tests.
- Run `go test ./internal/modules/dnd/... ./internal/framework/llm
./internal/cli`.
- Run `go test ./...`.
- Verify `rg -n 'default_profile: gemini-2-flash' internal/modules/dnd`
returns no matches.
- Run `git diff --check`.
### Completion Criteria
- All maintained D&D prompts use the application-owned logical profile ID.
- The fallback works without an operator profile and remains authoritatively
overridable by a matching valid operator definition.
## Stage 6: Introduce Module Execution-Class Metadata
### Goal
Make each production module's ability to use an LLM statically discoverable
without yet changing profile inheritance.
### Work
- Add `ExecutionClass contracts.ExecutionClass` to `pipeline.ModuleSpec` and
preserve it through normalization, cloning, catalogs, registries, JSON/debug
views, and lookup helpers.
- In this transitional stage only, allow an omitted execution class to
normalize to deterministic so existing test-only fixtures can be migrated in
Stage 7 without breaking the repository midway.
- Explicitly classify every production module:
- D&D scene chunking, every D&D extractor, and D&D NPC normalization as
`llm_backed`;
- all other current production input, chunk, merge, normalize, and output
modules as `deterministic`.
- Update production module specification tests and production catalog tests to
assert the semantic class alongside stage, artifact kind, and capabilities.
- Add catalog lookup support needed by later resolution to retrieve a selected
module's execution class by stage and key without constructing it.
- Do not implement pipeline-level profile inheritance or reject deterministic
profiles yet.
- Update `docs/internal/modules.md` and `docs/internal/dnd.md` to identify
execution class as registered module metadata, while noting only implemented
uses.
### Tests And Validation
- Run module registration/spec tests across generic, Seriatim, and D&D
families.
- Run `go test ./internal/framework/pipeline ./internal/modules/...`.
- Run `go test ./...` and `git diff --check`.
### Completion Criteria
- Every production module has an explicit correct execution class.
- Catalog consumers can retrieve that class without a concrete module
instance.
- Test-only omitted classes remain the only temporary compatibility behavior.
## Stage 7: Enforce Execution Metadata And Remove Runtime Probing
### Goal
Finish the execution-class contract so missing metadata cannot cause future
profile drift.
### Work
- Update every framework, CLI, and integration test module specification to
declare an explicit execution class appropriate to the fake behavior.
- Change module-spec validation so an empty or unsupported execution class is a
registration error. Remove the transitional deterministic default from
Stage 6.
- Replace the chunk runner's special `ChunkExecutionClassProvider` probe with
specification-derived behavior. Remove the now-redundant provider interface,
implementation methods, and tests when they have no remaining consumer.
- Ensure chunk producer provenance remains unchanged: it records a non-empty
effective binding profile for an LLM-backed chunker, while a deterministic
chunker records no profile. A profile selected only through the prompt
default remains represented by PromptKit's actual-profile manifest rather
than being invented as an explicit chunk binding.
- Review helper constructors and fixtures for opportunities to set execution
class once without obscuring the class under test. Do not introduce an
elaborate test-spec framework.
- Update internal documentation if the removal changes any described runtime
mechanics.
### Tests And Validation
- Add or retain focused registration tests for missing and invalid execution
classes.
- Retain chunk-plan provenance tests for LLM-backed and deterministic
chunkers.
- Run `go test ./internal/framework/pipeline ./internal/modules/...`.
- Run `go test ./...`, `go vet ./...`, and `git diff --check`.
### Completion Criteria
- No registered module specification relies on an implicit execution class.
- Pipeline metadata, not a concrete runtime type assertion, owns module
execution classification.
## Stage 8: Resolve Programmatic Pipeline Profile Defaults
### Goal
Implement profile inheritance and precedence inside the pipeline resolver
before exposing the field through YAML configuration.
### Work
- Add an optional trimmed `LLMProfile` field to
`pipeline.PipelineProfile`. Add a non-empty runtime override field to
`pipeline.ResolveOptions` so all precedence decisions occur in the resolver
rather than through pre-resolution mutation.
- After module selection, `--only` filtering, default validator-chain
selection, and validator compatibility resolution, apply effective profiles
to every selected input, chunk, extract, merge, normalize, output, and
validator binding according to the precedence in `promptkit.md`.
- Apply profiles only when the selected module or validator execution class is
`llm_backed`.
- Reject a binding-specific `llm_profile` on any deterministic module or
validator. Do not reject or inspect an unused pipeline default when no
selected LLM-backed binding consumes it.
- Leave an LLM-backed binding empty when no CLI, binding, or pipeline profile is
selected so PromptKit can use the prompt's `default_profile`.
- Store the effective values on resolved bindings before digest construction.
Do not add a second inheritance decision to execution.
- Ensure semantically equivalent repeated binding profiles and one inherited
default produce the same resolved pipeline digest. Ensure any changed
effective profile changes the digest.
- Do not modify file configuration or CLI parsing in this stage.
### Tests And Validation
- Add pipeline package tests for the complete precedence matrix:
runtime override; binding-specific exception; pipeline default; prompt
fallback; and deterministic bindings.
- Cover default and explicitly configured validator chains, all relevant stage
categories, `--only` lane selection, unused defaults, deterministic-profile
rejection, and semantic digest equivalence.
- Prefer table-driven package-level tests over assertions on private traversal
helpers.
- Run `go test ./internal/framework/pipeline` and `go test ./...`.
- Run `git diff --check`.
### Completion Criteria
- Programmatic pipelines resolve one canonical effective profile policy.
- Only LLM-backed resolved bindings can contain a profile.
- Runtime override, binding, pipeline, and prompt precedence is unambiguous and
digest-stable.
## Stage 9: Expose Pipeline Defaults Through Configuration And CLI
### Goal
Make the profile-default workflow available to operators while preserving
validation and override behavior.
### Work
- Add optional `pipelines.<id>.llm_profile` support to the version 4 file
configuration model. Use presence-aware decoding so an explicitly set blank
value is rejected, while omission remains valid.
- Preserve the field through file application, configuration cloning,
effective configuration, and programmatic profile copies without aliasing or
trimming drift.
- Remove `applyLLMProfileOverride`. Pass the CLI override through the resolver's
runtime-override input so deterministic bindings are never populated.
- Update effective profile-ID collection to cover every selected LLM-backed
module stage and LLM-backed validator, including future LLM-backed input and
output modules. Do not inspect deterministic or unselected profiles.
- Ensure `run`, `config validate --pipeline`, resume/checkpoint identity, and
relevant dry preflight paths all use the same resolved effective profiles.
- Preserve `--llm-profile` as the highest-precedence non-empty run-wide
override and preserve binding-specific profiles as exceptions when no CLI
override is present.
- Do not increment the configuration version.
- Update current configuration and CLI contracts in `docs/config.md` and
`docs/cli.md` in the same stage. Link to operations for the deployment
workflow rather than duplicating it prematurely.
### Tests And Validation
- Add file-config tests for omission, trimming, explicit blank rejection,
unknown-key behavior, cloning, and round-trip application.
- Add effective-config and CLI contract tests for precedence, LLM-only
application, inherited-profile inspection failure before factory execution,
`--only`, and digest changes.
- Retain offline operation and do not require credentials for
`config validate --pipeline`.
- Run `go test ./internal/core/config ./internal/framework/pipeline
./internal/cli`.
- Run `go test ./...`, `go vet ./...`, and `git diff --check`.
### Completion Criteria
- Operators can select `dnd-extraction` once per pipeline.
- Configuration and CLI paths share the resolver's precedence policy.
- Unknown effective profiles fail preflight, while deterministic and unused
profiles do not cause spurious inspection.
## Stage 10: Complete Operator Documentation, Examples, And Decision Record
### Goal
Make the implemented workflow understandable, copyable, and maintainable
without duplicating canonical facts.
### Work
- Create an ADR using the next sequential number for the durable decision to
use workload-oriented pipeline defaults with operator-overridable application
fallback profiles. Record context, decision, alternatives, and consequences;
do not turn the ADR into a field reference or implementation log.
- Complete `docs/config.md` as the canonical owner of profile-source fields,
`pipelines.<id>.llm_profile`, validation, and precedence.
- Complete `docs/operations.md` with an operator workflow that distinguishes
Notarius embedded prompts, Notarius fallback profiles, PromptKit built-ins,
and deployment filesystem profiles. Include production/development/local use
of the same `dnd-extraction` ID, credential handling, absolute-path guidance,
and the fact that current relative profile paths use the process working
directory rather than the configuration file's directory.
- Complete `docs/integrations/pkg-promptkit.md` with the v0.5.0 boundary,
prepared execution, inspection, fallback and ordinary source precedence,
optional provider controls, capacity adaptation, and compatibility policy.
- Update `docs/internal/configuration.md`, `docs/internal/pipeline.md`,
`docs/internal/cli.md`, `docs/internal/llm.md`, `docs/internal/modules.md`, and
`docs/internal/dnd.md` only for their owned implementation details. Link to
canonical configuration, operations, and upstream format contracts rather
than restating them.
- Keep exactly the existing two D&D configuration examples. Add
`llm_profile: dnd-extraction` to the minimal and complete pipelines and remove
the now-redundant model-named binding override from the complete example.
- Add one secret-free maintained operator profile at
`examples/profiles/dnd-extraction.yml`. It should be a complete valid profile
for the same logical ID and may mirror the embedded baseline; its purpose is
to demonstrate file ownership and format, not claim automatic environment
detection. Link it from the configuration and operations documentation.
- If the complete example selects the external profile file, use a path that
is valid for the documented repository-root invocation and explicitly note
the working-directory rule. Keep the minimal example dependent only on the
embedded fallback.
- Add or extend maintained-example validation so both configuration examples
and the profile YAML are checked without generation or credentials.
- Remove the now-implemented `Pipeline-Level LLM Profile Defaults` section from
`docs/roadmap/future.md`. Preserve the unrelated deterministic session and
concurrency items.
- Do not delete `promptkit.md` or this implementation plan during the feature
implementation; retire them only after post-implementation review.
### Tests And Validation
- Run maintained example/configuration tests and relevant CLI help/parser
tests.
- Run `go test ./...`.
- Run `rg -n 'gemini-2-flash' examples docs` and review every remaining match
for intentional model-policy or historical context.
- Run `rg -n 'v0\.3\.0|profileCheckPrompt|applyLLMProfileOverride' .` and resolve
stale production or current-documentation matches.
- Verify all new links and `git diff --check`.
### Completion Criteria
- Every current fact has one canonical documentation owner.
- Operators can distinguish and deploy all profile layers without reading Go
source.
- Both maintained configurations and the maintained external profile are valid,
secret-free, and tested offline.
- Implemented profile work no longer remains in `future.md`.
## Stage 11: Final Verification And Quality Review
### Goal
Verify the complete migration as one integrated change and correct only defects
or omissions found during that review.
### Work
- Review the final diff against every acceptance criterion in `promptkit.md`.
- Confirm provider-specific PromptKit types remain inside the LLM integration
boundary and D&D policy remains inside the D&D module family.
- Confirm execution and inspection receive identical ordinary, fallback, and
backend configuration.
- Confirm no paths, profile YAML, endpoints, credentials, or prepared handle
state leak into fingerprints or ordinary diagnostics.
- Confirm all production module specs have explicit correct execution classes
and every resolved deterministic binding is profile-free.
- Confirm prompt default, pipeline default, binding override, and CLI override
behavior through representative assembled configurations.
- Review tests for redundancy and remove obsolete synthetic-prompt,
runtime-probe, exact-hash, or duplicated upstream-behavior tests superseded by
stronger contract tests.
- Perform an optional manual D&D quality comparison if credentials and an
evaluation transcript are deliberately supplied. Record no private input or
credential material, and do not make this comparison a completion gate.
### Validation Commands
```sh
gofmt -w <changed-go-files>
go test ./...
go test -race ./internal/framework/llm ./internal/core/config ./internal/framework/pipeline ./internal/cli
go vet ./...
go build ./cmd/notarius
git diff --check
```
Also run focused stale-contract searches:
```sh
rg -n 'gitea.maximumdirect.net/eric/promptkit v0\.3\.0|PromptKit v0\.3\.0' .
rg -n 'default_profile: gemini-2-flash|profileCheckPrompt|applyLLMProfileOverride' internal docs examples
```
Review any matches rather than deleting intentional historical references
blindly.
### Completion Criteria
- All automated checks pass offline and without real credentials.
- The implemented behavior matches `promptkit.md` with no known architecture,
provenance, checkpoint, profile-precedence, or documentation gap.
- Any optional live evaluation is clearly separate from correctness testing.
## Open Questions
None. The roadmap decisions are sufficient to implement every stage without an
additional product or architecture choice.

View File

@@ -1,318 +0,0 @@
# PromptKit v0.5 Integration And LLM Profile Policy
## Purpose
This roadmap defines the target state for upgrading Notarius from PromptKit
v0.3.0 to v0.5.0 and adopting the upstream runtime and profile facilities that
directly improve Notarius. It also defines the application policy for stable,
domain-oriented LLM profile names, operator overrides, pipeline inheritance,
profile validation, provider defaults, checkpoint identity, and documentation.
The ordered work needed to reach this state belongs in
[the implementation plan](implementation.md). Current behavior remains defined
by the canonical documentation outside `docs/roadmap/` until the corresponding
work is implemented.
## Background
Notarius currently pins PromptKit v0.3.0. Its adapter prepares a request once
for debug material and then independently runs the original request, causing
PromptKit to prepare the same logical call a second time. The CLI validates an
explicit profile by preparing a synthetic prompt. PromptKit profile selection
can be repeated on individual module bindings or replaced for one invocation
with `--llm-profile`, but a configured pipeline cannot yet declare one inherited
profile policy.
PromptKit v0.4.0 and v0.5.0 add the upstream boundaries needed to improve these
areas:
- [v0.4.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/releases/v0.4.0.md)
adds opaque prepared executions, exact profile and prompt inspection, and a
typed backend-capacity error;
- [v0.5.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/releases/v0.5.0.md)
adds application fallback profile filesystems and stops sending unset
optional sampling controls as framework-selected provider values; and
- the [v0.5.0 format contract](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.5.0/docs/formats.md)
defines the resulting profile-source and execution-setting precedence.
A source-compatibility test of the current Notarius repository against
PromptKit v0.5.0 completed successfully. The work is therefore primarily an
intentional runtime and configuration migration rather than a repair for a
breaking Go API change.
## Goals
- Pin and document PromptKit v0.5.0 as Notarius's supported upstream contract.
- Execute the exact prepared request snapshot whose safe details are recorded
in Notarius debug material.
- Validate configured PromptKit profiles through the upstream inspection API
without synthetic prompts, provider calls, or credential-value access.
- Give Notarius an application-owned, operator-overridable
`dnd-extraction` profile fallback.
- Let a pipeline choose one default LLM profile without repeating that ID on
every LLM-backed binding.
- Apply profile inheritance and run-wide overrides only where the resolved
module or validator can use an LLM.
- Preserve accurate checkpoint invalidation, effective profile provenance,
redaction, cancellation, concurrency, and provider-neutral module contracts.
- Provide operators with one clear deployment pattern for production,
development, and local profile definitions.
## Target End State
### PromptKit Runtime Boundary
Notarius depends on PromptKit v0.5.0 and uses its public APIs rather than
reimplementing source or execution resolution.
For each structured completion, the adapter:
1. builds one PromptKit run request from the provider-neutral Notarius request;
2. calls `PrepareExecution` once;
3. immediately arranges an idempotent `Discard` for every unexecuted handle;
4. obtains credential-redacted `Details` for debug and response metadata; and
5. calls `RunPrepared` so generation uses that exact frozen snapshot.
The debug prompt and successful result therefore describe the same selected
profile, rendered messages, input bytes, session, output contract, and effective
settings even when a filesystem-backed source changes concurrently. PromptKit
handle types remain private to `internal/framework/llm`.
PromptKit admission failures continue to match Notarius's provider-neutral
`ErrLLMCapacityExceeded` contract. When PromptKit supplies a `CapacityError`,
the adapter obtains the normalized backend ID through `errors.As` and may add it
to safe application-owned diagnostics without parsing upstream error wording.
The backend ID does not become a provider-specific module contract.
### Optional Provider Controls
Notarius accepts PromptKit v0.5.0's new behavior for `temperature`,
`max_tokens`, and `top_p`: an unset setting is omitted from compatible provider
requests and the provider chooses its own default. Notarius does not restore
PromptKit's former implicit `top_p: 1` value globally.
An operator who requires a particular value specifies it in the selected
PromptKit profile. The application fallback described below intentionally
leaves these controls unset. A human-reviewed D&D extraction comparison should
be performed after the upgrade, but paid or nondeterministic model output is
not part of the default automated test suite.
### Profile Inspection
Pipeline-aware configuration validation uses `Engine.InspectProfile` for every
effective explicit profile ID. It verifies that the profile exists, parses and
validates, resolves its backend and target, and is compatible with the engine's
registered backends. It does not create a synthetic prompt, load prompt inputs,
contact a provider, or require credential values to exist in the validation
process environment.
Credential availability is execution-time state. PromptKit preparation still
enforces the selected profile's credential contract before generation. This
keeps `notarius config validate` useful in build and deployment validation
environments where secrets are deliberately absent.
PromptKit construction for inspection and execution uses one shared internal
profile-source and backend-option path. The CLI does not expose PromptKit public
types across the Notarius LLM boundary merely to perform inspection.
`InspectPrompt` is not adopted merely because it exists. It remains available
for a later, separately defined module-to-prompt interface preflight if a
concrete validation requirement justifies that additional contract.
### Application And Operator Profile Sources
Notarius embeds one ordinary PromptKit YAML profile with the stable ID
`dnd-extraction`. It is an application fallback registered through
`WithFallbackProfileFS`, is owned by the D&D module family, and initially
preserves the current effective D&D baseline:
- backend: PromptKit's built-in `openrouter` backend;
- model: `openai/gpt-5.6-luna`;
- reasoning effort: unset, allowing OpenAI's backend to apply its default of
`medium`;
- generation timeout: 240 seconds;
- service tier: `flex`; and
- no application-selected `temperature`, `max_tokens`, or `top_p`.
All maintained D&D LLM prompt definitions use `dnd-extraction` as their
`default_profile`. The ID communicates workload intent rather than a provider,
model, or environment. Changing the embedded fallback is an intentional
Notarius execution-policy change and participates in checkpoint identity.
Effective profile definitions resolve in PromptKit's order:
1. programmatic in-memory profiles used by tests or explicit consumers;
2. the operator source configured by `promptkit.profile_file` or
`promptkit.profile_dir`;
3. the Notarius application fallback source; and
4. PromptKit's embedded built-in catalog.
Only an absent ID falls through to the next source. A matching profile is a
complete definition: fields are not merged with a lower-precedence definition,
and a malformed matching operator profile fails rather than silently selecting
the application fallback.
Production, development, and local deployments should normally provide
different complete definitions for the same `dnd-extraction` ID. An operator
source is optional because the application fallback keeps the maintained D&D
workflow usable, but a deployment that needs an intentional model or backend
policy should configure its own definition.
### Domain Ownership And Asset Assembly
The D&D fallback profile remains under `internal/modules/dnd` and is registered
by the D&D registrar, consistent with ADR-0004. Generic LLM plumbing knows how
to collect and flatten application fallback profile filesystems but contains no
D&D model or policy knowledge.
The shared asset registry detects invalid roots, unreadable sources, and
duplicate flattened paths. PromptKit remains responsible for strict profile
YAML parsing, duplicate profile-ID detection, source precedence, and effective
target resolution. The same assembled fallback source is supplied to runtime
execution and CLI profile inspection.
### Explicit Module Execution Metadata
Every registered input, chunk, extract, merge, normalize, and output module
declares one required execution class: `deterministic` or `llm_backed`.
Validator registrations continue to declare the same distinction through their
validator specifications.
The registered specification is authoritative for configuration resolution.
Current production classifications are:
- the D&D scene chunker, all D&D extractors, and the D&D NPC normalizer are
LLM-backed;
- the Seriatim input adapter, generic chunker, all current mergers, all other
current normalizers, and the JSON output encoder are deterministic; and
- current validators retain their declared classifications.
Missing or unsupported execution metadata is a registration error. Explicitly
assigning `llm_profile` to a deterministic module or validator is a pipeline
resolution error. The framework does not infer execution class by inspecting
domain package names or concrete implementation types at runtime.
The module specification replaces the chunk runner's special runtime
execution-class probe. Effective resolved bindings already express the result:
only LLM-backed bindings may retain a non-empty profile.
### Pipeline-Level Profile Default
Configuration version 4 gains one optional non-empty pipeline field:
```yaml
pipelines:
dnd-session:
llm_profile: dnd-extraction
```
No configuration-version increment is required because the field is additive
and existing files remain valid. An explicitly present blank value is invalid.
For every selected LLM-backed module and validator, the effective profile uses
this precedence:
1. non-empty run-wide `--llm-profile` override;
2. binding-specific `llm_profile`;
3. pipeline-level `llm_profile`; and
4. the prompt definition's `default_profile`, represented by an empty effective
Notarius binding profile.
The run-wide override and inherited pipeline default never attach to a
deterministic binding. Binding-specific exceptions remain available when one
operation needs a different cost, latency, quality, backend, or reasoning
policy.
Inheritance is resolved after module and validator selection, including
`--only` lane filtering, but before effective-pipeline validation, digest
construction, explicit-profile inspection, checkpoint construction,
preparation, execution, or provenance capture. Only profiles used by selected
LLM-backed bindings are inspected. An unused pipeline default in a pipeline
with no selected LLM-backed work does not require an otherwise unused profile
to exist.
The resolved pipeline contains effective binding profiles rather than a second
runtime inheritance mechanism. Two pipelines that differ only by spelling the
same effective policy once as a pipeline default and once on every LLM-backed
binding have the same semantic resolved digest. Changing an effective profile
changes the digest and applicable checkpoint identity.
### Provenance And Checkpoints
The PromptKit profile-source checkpoint fingerprint covers:
- the PromptKit v0.5.0 built-in profile catalog identity;
- exact application fallback profile asset content; and
- exact configured operator profile YAML content, when present.
The existing local-backend target fingerprint remains separate and continues
to exclude scheduling-only concurrency limits. Fingerprints contain hashes and
stable markers, not profile contents, filesystem paths, endpoints, credentials,
or other secrets.
Changing the PromptKit version, application fallback, operator profile, or
effective pipeline profile makes incompatible LLM checkpoints ineligible for
reuse. The dependency upgrade is expected to invalidate checkpoints produced
under v0.3.0.
Successful run manifests continue to record only profiles actually selected by
PromptKit, including their effective model, backend, and reasoning metadata.
Debug output reports the same effective execution snapshot used for generation.
### Operator Documentation And Examples
Canonical documentation clearly distinguishes:
- Notarius prompt and schema assets embedded in the application;
- Notarius application fallback profiles embedded in the application;
- PromptKit's own embedded built-in profiles; and
- operator profile files on the deployment filesystem.
The configuration reference owns the pipeline field, profile-source fields,
validation rules, and precedence. Operations owns deployment layout, working
directory behavior, credentials, and environment-specific profile management.
The PromptKit integration document owns the pinned upstream contract and
source-precedence boundary. Internal documents describe asset registration,
resolution, inspection, prepared execution, fingerprinting, and tests without
duplicating user-facing field definitions.
The maintained examples continue to include only the minimal and complete D&D
configurations. They use the stable `dnd-extraction` policy, and one maintained
PromptKit profile file under `examples/` demonstrates an operator override.
Examples remain secret-free and are validated without live provider calls.
## Out Of Scope
- Implementing the separate deterministic prompt-session identity roadmap
item.
- Changing the default `concurrency.total_llm` value; PromptKit's retained
OpenRouter capacity of 16 remains relevant to that separate item.
- Adding model evaluation as a deterministic or CI correctness gate.
- Automatically selecting production, development, or local environments.
Deployment configuration chooses the operator profile source.
- Profile inheritance, partial profile merging, or cross-profile aliases.
- Exposing PromptKit types to modules, validators, durable output contracts, or
public configuration structures.
- Adopting `InspectPrompt` without a separately justified prompt-interface
validation contract.
## Acceptance Criteria
- Notarius builds and its offline test suite passes with PromptKit v0.5.0.
- Every structured completion executes the exact snapshot used for safe debug
prompt details.
- Profile preflight uses profile inspection and no synthetic prompt.
- The embedded `dnd-extraction` fallback resolves without an operator source,
and a matching valid operator profile replaces it completely.
- Every production module has explicit, correct execution metadata.
- Pipeline, binding, CLI, and prompt-default precedence behaves as defined for
modules and validators, while deterministic bindings remain profile-free.
- Effective profiles participate in pipeline digests, profile inspection,
checkpoint identity, debug records, and run provenance at the appropriate
boundaries.
- The dependency and application fallback changes invalidate incompatible old
checkpoints without exposing profile or credential content.
- Canonical documentation and maintained examples accurately describe and
exercise the implemented operator workflow.
- Default tests remain deterministic, offline, credential-free, and focused on
Notarius-owned behavior rather than duplicating PromptKit's upstream suite.

View File

@@ -36,10 +36,13 @@ pipelines:
window_units: 3
lanes:
- item-events
- locations
- location-occurrences
- npcs
- spells
- combat-turns
- npc-interactions
- enemy-events
steps:
# Establish session-wide reference artifacts alongside independent item events.
- id: describe-session
@@ -58,6 +61,14 @@ pipelines:
normalize:
module: dnd/npcs
retries: 2
locations:
extract:
module: dnd/locations
retries: 2
merge: appendorder
normalize:
module: dnd/locations
retries: 2
scene-descriptions:
extract:
module: dnd/scene-descriptions
@@ -65,9 +76,13 @@ pipelines:
merge: appendorder
normalize: dnd/scene-descriptions
- id: extract-events
# Accepted NPC grounding and scene-description eligibility artifacts are
# supplied in memory to their compatible consumers in this step.
# Accepted registry artifacts and scene-description eligibility artifacts
# are supplied in memory to their compatible consumers in this step.
references:
locations:
artifact:
step: describe-session
lane: locations
npcs:
artifact:
step: describe-session
@@ -101,3 +116,34 @@ pipelines:
retries: 2
merge: appendorder
normalize: dnd/npc-interactions
location-occurrences:
extract:
module: dnd/location-occurrences
retries: 2
merge: appendorder
normalize: dnd/location-occurrences
- id: track-enemies
references:
npcs:
artifact:
step: describe-session
lane: npcs
scene_descriptions:
artifact:
step: describe-session
lane: scene-descriptions
combat_turns:
artifact:
step: extract-events
lane: combat-turns
npc_interactions:
artifact:
step: extract-events
lane: npc-interactions
artifacts:
enemy-events:
extract:
module: dnd/enemy-events
retries: 2
merge: appendorder
normalize: dnd/enemy-events

View File

@@ -249,10 +249,24 @@ func TestRunAutoReusesPlanWhenRunInputsChange(t *testing.T) {
}
harness.mu.Lock()
chunkCalls := harness.chunkCalls
sessions := append([]string(nil), harness.sessionIDs...)
harness.mu.Unlock()
if chunkCalls != 1 {
t.Fatalf("chunk calls across changed run inputs = %d, want 1", chunkCalls)
}
rawInput, err := os.ReadFile(roots.input)
if err != nil {
t.Fatal(err)
}
wantSessionID, err := resolvePromptSessionID("", "test/input", rawInput)
if err != nil {
t.Fatal(err)
}
for _, sessionID := range sessions {
if sessionID != wantSessionID {
t.Fatalf("session IDs across reference changes = %#v, want %q", sessions, wantSessionID)
}
}
assertFile(t, filepath.Join(roots.plans, strings.TrimPrefix(stateTestDigest, "sha256:"), "plan.json"))
assertAnyFile(t, roots.output)
}

View File

@@ -0,0 +1,340 @@
package cli
import (
"context"
"encoding/json"
"errors"
"fmt"
"os"
"path/filepath"
"reflect"
"strings"
"sync"
"testing"
"time"
"gitea.maximumdirect.net/eric/notarius/internal/core/artifacts"
"gitea.maximumdirect.net/eric/notarius/internal/core/config"
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
"gitea.maximumdirect.net/eric/notarius/internal/framework/evidencecontext"
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/chunk/scenes"
locationoccurrencecodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/locationoccurrences"
locationcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/locations"
combat "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/combatturns"
enemyevents "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/enemyevents"
itemevents "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/itemevents"
locationoccurrences "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/locationoccurrences"
locations "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/locations"
npcinteractions "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/npcinteractions"
npcs "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/npcs"
scenedescriptions "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/scenedescriptions"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/spells"
enemyeventnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/enemyevents"
locationnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/locations"
npcnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/npcs"
)
func TestProductionEnemyEventConfigurationResolvesGeneratedHandoffs(t *testing.T) {
components := productionTestComponents(t)
cfg := loadMaintainedExample(t, repositoryPath("examples", "dnd-complete.config.yml"))
effective, err := cfg.Resolve(resolveInputForMaintainedExample(components, "dnd-session"))
if err != nil {
t.Fatalf("Resolve() error = %v, want nil", err)
}
materialized, _, err := pipeline.MaterializeReferences(effective.ResolvedPipeline, catalogFromRegistries(components.registries), pipeline.ReferenceMaterializationOptions{
ConfigPath: repositoryPath("examples", "dnd-complete.config.yml"),
WorkingDir: repositoryPath("examples"),
})
if err != nil {
t.Fatalf("MaterializeReferences() error = %v, want nil", err)
}
lane := referenceContractLane(t, materialized, "enemy-events")
if lane.ArtifactKind != dnd.EnemyEventListKind || lane.Extract.Module != enemyevents.Key || lane.Extract.Retries != 2 || lane.Merge.Module != pipeline.DefaultMergeModule || lane.Normalize.Module != enemyeventnormalize.Key {
t.Fatalf("enemy event lane = %#v, want typed production composition", lane)
}
for slot, want := range map[string]struct{ step, lane string }{
"npcs": {step: "describe-session", lane: "npcs"},
"scene_descriptions": {step: "describe-session", lane: "scene-descriptions"},
"combat_turns": {step: "extract-events", lane: "combat-turns"},
"npc_interactions": {step: "extract-events", lane: "npc-interactions"},
} {
binding, found := generatedReferenceBinding(lane.ExtractReferences.Bindings, slot)
if !found || binding.Artifact.Step != want.step || binding.Artifact.Lane != want.lane {
t.Fatalf("enemy event %s reference = %#v, want generated %s/%s artifact", slot, binding, want.step, want.lane)
}
}
if binding, found := generatedReferenceBinding(lane.NormalizeReferences.Bindings, "npcs"); !found || binding.Artifact.Step != "describe-session" || binding.Artifact.Lane != "npcs" {
t.Fatalf("enemy event normalizer NPC reference = %#v, want generated NPC artifact", binding)
}
catalog := catalogFromRegistries(components.registries)
extractSpec, ok := catalog.Extractors.Spec(enemyevents.Key)
if !ok || !reflect.DeepEqual(extractSpec.Requires, []string{"chunks", "source.transcript"}) || !reflect.DeepEqual(extractSpec.Provides, []string{"dnd.enemy_events"}) {
t.Fatalf("enemy event extractor spec = %#v, want source and artifact capabilities", extractSpec)
}
normalizeSpec, ok := catalog.Normalizers.SpecForArtifact(enemyeventnormalize.Key, dnd.EnemyEventListKind)
if !ok || !reflect.DeepEqual(normalizeSpec.Requires, []string{"merged"}) || !reflect.DeepEqual(normalizeSpec.Provides, []string{"normalized"}) {
t.Fatalf("enemy event normalizer spec = %#v, want merged/normalized capabilities", normalizeSpec)
}
for _, slot := range []string{"npcs", "scene_descriptions", "combat_turns", "npc_interactions"} {
if !hasReferenceSlot(extractSpec.ReferenceSlots, slot) {
t.Fatalf("enemy event extractor slots = %#v, want %q", extractSpec.ReferenceSlots, slot)
}
}
if !hasReferenceSlot(normalizeSpec.ReferenceSlots, "npcs") {
t.Fatalf("enemy event normalizer slots = %#v, want NPC registry", normalizeSpec.ReferenceSlots)
}
profile := cfg.Pipelines["dnd-session"]
profile.Steps[2].References["npcs"] = pipeline.GeneratedReference("track-enemies", "enemy-events")
cfg.Pipelines["dnd-session"] = profile
if _, err := cfg.Resolve(resolveInputForMaintainedExample(components, "dnd-session")); err == nil || !strings.Contains(err.Error(), "earlier step") {
t.Fatalf("Resolve() error = %v, want future generated-reference rejection", err)
}
}
func TestMaintainedCompleteExampleProducesEnemyEventsThroughGeneratedHandoffs(t *testing.T) {
t.Chdir(repositoryPath())
outputRoot := filepath.Join(t.TempDir(), "output")
configPath := completeExampleConfigWithTemporaryCache(t)
client := &enemyEventLLMClient{}
options := productionCLIOptions(t)
options.Now = func() time.Time { return time.Unix(1700000000, 0).UTC() }
options.RunIDGenerator = func(time.Time) (string, error) { return productionRunID, nil }
options.UserCacheDir = func() (string, error) { return "", errors.New("user cache must not be used") }
options.LLMClientFactory = func(context.Context, config.Config, string, LLMRuntimeOverrides) (contracts.StructuredLLMClient, []artifacts.LLMProfileManifest, error) {
return client, nil, nil
}
var stdout, stderr strings.Builder
code := RunWithOptions([]string{
"run", "dnd-session",
"--config", configPath,
"--input", repositoryPath("examples", "dnd-complete-transcript.json"),
"--chunk_cache", "bypass", "--output-dir", outputRoot, "--session-id", "enemy-event-session",
}, &stdout, &stderr, options)
if code != 0 {
t.Fatalf("code=%d stdout=%q stderr=%q", code, stdout.String(), stderr.String())
}
runRoot := filepath.Join(outputRoot, productionRunID)
index := readProductionJSON[exampleOutputIndex](t, filepath.Join(runRoot, "index.json"))
var enemyOutput exampleOutputIndexEntry
var locationOutput, occurrenceOutput exampleOutputIndexEntry
for _, entry := range index.OutputFiles {
switch entry.LaneID {
case "enemy-events":
enemyOutput = entry
case "locations":
locationOutput = entry
case "location-occurrences":
occurrenceOutput = entry
}
}
if enemyOutput.File != "lanes/enemy-events.json" || enemyOutput.SchemaID != "notarius.dnd.enemy_events" || enemyOutput.SchemaVersion != "v1" {
t.Fatalf("enemy event output = %#v, want typed enemy-event JSON", enemyOutput)
}
value := readProductionJSON[dnd.EnemyEventList](t, filepath.Join(runRoot, enemyOutput.File))
if len(value.Events) != 1 || value.Events[0].Name != "Kesh" || value.Events[0].Kind != dnd.EnemyEventKindFled || len(value.Events[0].SourceRefs) != 1 || value.Events[0].SourceRefs[0].SourceID != "session-ravenfall" || value.Events[0].SourceRefs[0].StartUnitID != 10 {
t.Fatalf("enemy event artifact = %#v, want source-linked Kesh fleeing event", value)
}
if locationOutput.File != "lanes/locations.json" || locationOutput.SchemaID != locationcodec.SchemaID || locationOutput.SchemaVersion != locationcodec.SchemaVersion {
t.Fatalf("location output = %#v, want typed location registry JSON", locationOutput)
}
locationsValue := readProductionJSON[dnd.LocationList](t, filepath.Join(runRoot, locationOutput.File))
if len(locationsValue.Locations) != 2 || locationsValue.Locations[0].Name != "Moon Gate" || locationsValue.Locations[1].Name != "Moon Gate" || locationsValue.Locations[0].ID == locationsValue.Locations[1].ID {
t.Fatalf("location registry = %#v, want distinct source-grounded identities for same-name locations", locationsValue)
}
if occurrenceOutput.File != "lanes/location-occurrences.json" || occurrenceOutput.SchemaID != locationoccurrencecodec.SchemaID || occurrenceOutput.SchemaVersion != locationoccurrencecodec.SchemaVersion {
t.Fatalf("location occurrence output = %#v, want typed occurrence JSON", occurrenceOutput)
}
occurrencesValue := readProductionJSON[dnd.LocationOccurrenceList](t, filepath.Join(runRoot, occurrenceOutput.File))
if len(occurrencesValue.Occurrences) != 2 || occurrencesValue.Occurrences[0].LocationID == occurrencesValue.Occurrences[1].LocationID || occurrencesValue.Occurrences[0].Name != "Moon Gate" || occurrencesValue.Occurrences[1].Name != "Moon Gate" {
t.Fatalf("location occurrences = %#v, want source-grounded references to distinct registry identities", occurrencesValue)
}
evidence := readProductionJSON[evidencecontext.Document](t, filepath.Join(runRoot, "evidence-context.json"))
for _, laneID := range []string{"enemy-events", "locations", "location-occurrences"} {
if !containsString(evidence.SelectedLanes, laneID) || !evidenceHasLane(evidence, laneID) {
t.Fatalf("evidence context = %#v, want direct %s evidence", evidence, laneID)
}
}
requests := client.requestsFor(enemyevents.PromptID)
if len(requests) != 1 {
t.Fatalf("enemy event requests = %#v, want only the combat scene request", requests)
}
request := requests[0]
if request.SessionID != "enemy-event-session" {
t.Fatalf("enemy event session = %q, want shared session", request.SessionID)
}
for slot, required := range map[string]string{
"npcs": "Kesh",
"combat_turns": "Kesh",
"npc_interactions": "Kesh",
} {
input, ok := request.Inputs[slot]
if !ok || !strings.Contains(string(input.Content), required) || strings.Contains(string(input.Content), "source_refs") || strings.Contains(string(input.Content), "start_unit_id") {
t.Fatalf("enemy event %s prompt input = %q, want compact source-free grounding", slot, input.Content)
}
}
locationRequests := client.requestsFor(locationoccurrences.PromptID)
if len(locationRequests) != 2 {
t.Fatalf("location occurrence requests = %#v, want one request per scene", locationRequests)
}
for _, request := range locationRequests {
registryInput := request.Inputs["locations"]
if !strings.Contains(string(registryInput.Content), "Moon Gate") || !strings.Contains(string(registryInput.Content), `"id"`) || strings.Contains(string(registryInput.Content), "source_refs") {
t.Fatalf("location occurrence registry input = %q, want source-free ID grounding", registryInput.Content)
}
}
}
func completeExampleConfigWithTemporaryCache(t *testing.T) string {
t.Helper()
content, err := os.ReadFile(repositoryPath("examples", "dnd-complete.config.yml"))
if err != nil {
t.Fatal(err)
}
cacheRoot := t.TempDir()
updated := strings.Replace(string(content), "directory: ./notarius-cache/chunk-plans", fmt.Sprintf("directory: %q", filepath.Join(cacheRoot, "chunk-plans")), 1)
updated = strings.Replace(updated, "directory: ./notarius-cache/checkpoints", fmt.Sprintf("directory: %q", filepath.Join(cacheRoot, "checkpoints")), 1)
for relative, absolute := range map[string]string{
"./dnd-party.txt": repositoryPath("examples", "dnd-party.txt"),
"./dnd-glossary.txt": repositoryPath("examples", "dnd-glossary.txt"),
"./dnd-spell-catalog.json": repositoryPath("examples", "dnd-spell-catalog.json"),
} {
updated = strings.ReplaceAll(updated, relative, fmt.Sprintf("%q", absolute))
}
path := filepath.Join(t.TempDir(), "dnd-complete.config.yml")
if err := os.WriteFile(path, []byte(updated), 0o600); err != nil {
t.Fatal(err)
}
return path
}
type enemyEventLLMClient struct {
mu sync.Mutex
requests []contracts.StructuredCompletionRequest
}
func (client *enemyEventLLMClient) CompleteStructured(ctx context.Context, request contracts.StructuredCompletionRequest, output any) (contracts.StructuredCompletionResponse, error) {
if err := ctx.Err(); err != nil {
return contracts.StructuredCompletionResponse{}, err
}
combatScene := strings.Contains(string(request.Inputs["transcript"].Content), "Roll initiative")
var content []byte
switch request.PromptID {
case scenes.PromptID:
content = []byte(`{"scenes":[{"start_unit_id":1,"end_unit_id":6},{"start_unit_id":7,"end_unit_id":11}]}`)
case npcs.PromptID:
if combatScene {
content = []byte(`{"npcs":[{"name":"Kesh","source_refs":[{"start_unit_id":7,"end_unit_id":7}]}]}`)
} else {
content = []byte(`{"npcs":[]}`)
}
case npcnormalize.PromptID:
content = []byte(`{"duplicate_groups":[]}`)
case scenedescriptions.PromptID:
kind, title := "narrative", "Arrival"
if combatScene {
kind, title = "combat", "Raiders attack"
}
content = []byte(fmt.Sprintf(`{"kind":%q,"title":%q,"summary":"session scene"}`, kind, title))
case locations.PromptID:
unitID := 1
if combatScene {
unitID = 7
}
content = []byte(fmt.Sprintf(`{"locations":[{"name":"Moon Gate","source_refs":[{"start_unit_id":%d,"end_unit_id":%d}]}]}`, unitID, unitID))
case locationnormalize.PromptID:
content = []byte(`{"duplicate_groups":[]}`)
case spells.PromptID:
content = []byte(`{"spell_casts":[]}`)
case itemevents.PromptID:
content = []byte(`{"events":[]}`)
case combat.PromptID:
content = []byte(`{"combat_turns":[{"actor":"Kesh","turn_kind":"turn","source_refs":[{"start_unit_id":8,"end_unit_id":8}]}]}`)
case npcinteractions.PromptID:
if combatScene {
content = []byte(`{"interactions":[{"name":"Kesh","kind":"combat_opponent","source_refs":[{"start_unit_id":7,"end_unit_id":7}]}]}`)
} else {
content = []byte(`{"interactions":[]}`)
}
case locationoccurrences.PromptID:
var registry struct {
Locations []struct {
ID string `json:"id"`
} `json:"locations"`
}
if err := json.Unmarshal(request.Inputs["locations"].Content, &registry); err != nil {
return contracts.StructuredCompletionResponse{}, fmt.Errorf("decode generated location registry: %w", err)
}
if len(registry.Locations) == 0 {
return contracts.StructuredCompletionResponse{}, fmt.Errorf("generated location registry has no locations")
}
unitID := 1
locationID := registry.Locations[0].ID
if combatScene {
unitID = 7
if len(registry.Locations) > 1 {
locationID = registry.Locations[1].ID
}
}
content = []byte(fmt.Sprintf(`{"occurrences":[{"location_id":%q,"name":"Moon Gate","kind":"visited","source_refs":[{"start_unit_id":%d,"end_unit_id":%d}]}]}`, locationID, unitID, unitID))
case enemyevents.PromptID:
content = []byte(`{"events":[{"name":"Kesh","kind":"fled","source_refs":[{"start_unit_id":10,"end_unit_id":10}]}]}`)
default:
return contracts.StructuredCompletionResponse{}, fmt.Errorf("unexpected prompt %q", request.PromptID)
}
if err := json.Unmarshal(content, output); err != nil {
return contracts.StructuredCompletionResponse{}, fmt.Errorf("populate fake structured target: %w", err)
}
client.mu.Lock()
client.requests = append(client.requests, request)
client.mu.Unlock()
return contracts.StructuredCompletionResponse{Content: content, Provider: "test", Model: "deterministic", ProfileID: request.ProfileID}, nil
}
func (client *enemyEventLLMClient) requestsFor(promptID string) []contracts.StructuredCompletionRequest {
client.mu.Lock()
defer client.mu.Unlock()
var requests []contracts.StructuredCompletionRequest
for _, request := range client.requests {
if request.PromptID == promptID {
requests = append(requests, request)
}
}
return requests
}
func containsString(values []string, want string) bool {
for _, value := range values {
if value == want {
return true
}
}
return false
}
func evidenceHasLane(value evidencecontext.Document, laneID string) bool {
for _, context := range value.Contexts {
for _, reference := range context.EvidenceRefs {
if reference.LaneID == laneID {
return true
}
}
}
return false
}
func generatedReferenceBinding(bindings []pipeline.ReferenceBinding, slotName string) (pipeline.ReferenceBinding, bool) {
for _, binding := range bindings {
if binding.SlotName == slotName && binding.Artifact != nil {
return binding, true
}
}
return pipeline.ReferenceBinding{}, false
}

View File

@@ -15,6 +15,10 @@ import (
"gitea.maximumdirect.net/eric/notarius/internal/framework/contracts"
"gitea.maximumdirect.net/eric/notarius/internal/framework/pipeline"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
locationoccurrenceextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/locationoccurrences"
locationextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/locations"
locationoccurrencenormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/locationoccurrences"
locationnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/locations"
spellnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/spells"
"gitea.maximumdirect.net/eric/notarius/internal/modules/seriatim/input/transcript"
)
@@ -24,6 +28,9 @@ func TestMaintainedExamplesLoadResolveAndList(t *testing.T) {
for _, example := range maintainedExampleFiles(t) {
t.Run(example.name, func(t *testing.T) {
cfg := loadMaintainedExample(t, example.path)
if example.name == "complete" && (cfg.Concurrency.TotalLLM != 2 || cfg.Concurrency.StageWorkers["extract"] != 2) {
t.Fatalf("complete example concurrency = %#v, want explicit limits of 2", cfg.Concurrency)
}
raw, err := os.ReadFile(example.transcriptPath)
if err != nil {
t.Fatalf("read maintained transcript %q: %v", example.transcriptPath, err)
@@ -51,8 +58,30 @@ func TestMaintainedExamplesLoadResolveAndList(t *testing.T) {
t.Fatalf("materialize maintained example references for %q: %v", pipelineID, err)
}
if example.name == "complete" {
if got := exampleStepLaneIDs(materialized); strings.Join(got, "|") != "describe-session:item-events,npcs,scene-descriptions|extract-events:combat-turns,npc-interactions,spells" {
t.Fatalf("complete example steps and lanes = %v, want every D&D extractor in the documented two-step composition", got)
if got := exampleStepLaneIDs(materialized); strings.Join(got, "|") != "describe-session:item-events,locations,npcs,scene-descriptions|extract-events:combat-turns,location-occurrences,npc-interactions,spells|track-enemies:enemy-events" {
t.Fatalf("complete example steps and lanes = %v, want the documented D&D extractor composition", got)
}
locationLane := referenceContractLane(t, materialized, "locations")
if locationLane.ArtifactKind != dnd.LocationListKind || locationLane.Extract.Module != locationextract.Key || locationLane.Extract.Retries != 2 || locationLane.Merge.Module != pipeline.DefaultMergeModule || locationLane.Normalize.Module != locationnormalize.Key || locationLane.Normalize.Retries != 2 {
t.Fatalf("location lane = %#v, want typed registry composition", locationLane)
}
occurrenceLane := referenceContractLane(t, materialized, "location-occurrences")
if occurrenceLane.ArtifactKind != dnd.LocationOccurrenceListKind || occurrenceLane.Extract.Module != locationoccurrenceextract.Key || occurrenceLane.Extract.Retries != 2 || occurrenceLane.Merge.Module != pipeline.DefaultMergeModule || occurrenceLane.Normalize.Module != locationoccurrencenormalize.Key {
t.Fatalf("location occurrence lane = %#v, want typed occurrence composition", occurrenceLane)
}
for _, target := range []pipeline.ResolvedReferenceTarget{occurrenceLane.ExtractReferences, occurrenceLane.NormalizeReferences} {
binding, found := generatedReferenceBinding(target.Bindings, "locations")
if !found || binding.Artifact.Step != "describe-session" || binding.Artifact.Lane != "locations" {
t.Fatalf("location occurrence %s reference = %#v, want generated location registry", target.Stage, binding)
}
}
for _, slot := range []string{"party", "glossary"} {
if len(occurrenceLane.ExtractReferences.ReferenceSet.Slots[slot].Items) != 1 {
t.Fatalf("location occurrence extractor %s reference was not materialized: %#v", slot, occurrenceLane.ExtractReferences)
}
if _, found := occurrenceLane.NormalizeReferences.ReferenceSet.Slots[slot]; found {
t.Fatalf("location occurrence normalizer unexpectedly consumes %s: %#v", slot, occurrenceLane.NormalizeReferences)
}
}
spellLane := referenceContractLane(t, materialized, "spells")
if len(spellLane.ExtractReferences.ReferenceSet.Slots["spell_catalog"].Items) != 1 ||
@@ -68,6 +97,18 @@ func TestMaintainedExamplesLoadResolveAndList(t *testing.T) {
t.Fatalf("item event lane unexpectedly depends on generated scene descriptions: %#v", itemEventLane)
}
}
enemyEventLane := referenceContractLane(t, materialized, "enemy-events")
for slot, want := range map[string]struct{ step, lane string }{
"npcs": {step: "describe-session", lane: "npcs"},
"scene_descriptions": {step: "describe-session", lane: "scene-descriptions"},
"combat_turns": {step: "extract-events", lane: "combat-turns"},
"npc_interactions": {step: "extract-events", lane: "npc-interactions"},
} {
binding, found := generatedReferenceBinding(enemyEventLane.ExtractReferences.Bindings, slot)
if !found || binding.Artifact.Step != want.step || binding.Artifact.Lane != want.lane {
t.Fatalf("enemy event %s reference = %#v, want generated %s/%s artifact", slot, binding, want.step, want.lane)
}
}
}
}
var stdout, stderr strings.Builder
@@ -242,6 +283,12 @@ func TestMaintainedMalformedInputOnlyRecordsDebugFailureWhenRequested(t *testing
if report.Succeeded || report.PipelineID != "dnd-session" {
t.Fatalf("failure report = %#v, want failed dnd-session report", report)
}
invocation := readProductionJSON[debugbundle.Invocation](t, filepath.Join(bundle, "summary", "invocation.json"))
manifest := readProductionJSON[artifacts.RunManifest](t, filepath.Join(bundle, "summary", "run-manifest.json"))
manifestSession, found := manifest.Metadata["session_id"]
if invocation.SessionID == "" || !found || manifestSession != invocation.SessionID {
t.Fatalf("failed run sessions: invocation=%q manifest=%#v metadata=%#v", invocation.SessionID, manifestSession, manifest.Metadata)
}
})
}
}

View File

@@ -29,12 +29,15 @@ import (
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/chunk/scenes"
combatcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/combatturns"
enemyeventcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/enemyevents"
itemeventcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/itemevents"
spellcodec "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/codec/spells"
combatextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/combatturns"
enemyeventextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/enemyevents"
itemeventextract "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/itemevents"
"gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/extract/spells"
combatnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/combatturns"
enemyeventnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/enemyevents"
itemeventnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/itemevents"
spellnormalize "gitea.maximumdirect.net/eric/notarius/internal/modules/dnd/normalize/spells"
"gitea.maximumdirect.net/eric/notarius/internal/modules/generic/normalize/noop"
@@ -47,9 +50,9 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
assertProductionContains(t, "inputs", registries.Inputs.RegisteredKeys(), []string{"seriatim"})
assertProductionContains(t, "chunkers", registries.Chunkers.RegisteredKeys(), []string{"dnd/scenes", "generic"})
assertProductionContains(t, "extractors", registries.Extractors.RegisteredKeys(), []string{"dnd/spells", "dnd/npcs", combatextract.Key, itemeventextract.Key})
assertProductionContains(t, "extractors", registries.Extractors.RegisteredKeys(), []string{"dnd/spells", "dnd/npcs", combatextract.Key, itemeventextract.Key, enemyeventextract.Key})
assertProductionContains(t, "mergers", registries.Mergers.RegisteredKeys(), []string{"appendorder"})
assertProductionContains(t, "normalizers", registries.Normalizers.RegisteredKeys(), []string{"noop", spellnormalize.Key, "dnd/npcs", combatnormalize.Key, itemeventnormalize.Key})
assertProductionContains(t, "normalizers", registries.Normalizers.RegisteredKeys(), []string{"noop", spellnormalize.Key, "dnd/npcs", combatnormalize.Key, itemeventnormalize.Key, enemyeventnormalize.Key})
assertProductionContains(t, "outputs", registries.Outputs.RegisteredKeys(), []string{"json"})
assertProductionContains(t, "validators", registries.Validators.RegisteredKeys(), []string{
"extract/dnd/spells/catalog",
@@ -64,17 +67,23 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
"extract/dnd/item-events/source_refs",
"extract/dnd/item-events/source_relatedness",
"normalize/dnd/item-events/invariants",
"extract/dnd/enemy-events/shape",
"extract/dnd/enemy-events/engagements",
"extract/dnd/enemy-events/source_refs",
"extract/dnd/enemy-events/source_relatedness",
"normalize/dnd/enemy-events/invariants",
"generic/always_accept",
"generic/always_reject",
"generic/valid_json",
"generic/valid_json_schema",
})
assertProductionContains(t, "artifact codec kinds", registries.ArtifactCodecs.RegisteredKinds(), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind, dnd.ItemEventListKind})
assertProductionContains(t, "merger variants", registries.Mergers.RegisteredArtifactKinds(pipeline.DefaultMergeModule), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind, dnd.ItemEventListKind})
assertProductionContains(t, "normalizer variants", registries.Normalizers.RegisteredArtifactKinds(pipeline.DefaultNormalizeModule), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind, dnd.ItemEventListKind})
assertProductionContains(t, "artifact codec kinds", registries.ArtifactCodecs.RegisteredKinds(), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind, dnd.ItemEventListKind, dnd.EnemyEventListKind})
assertProductionContains(t, "merger variants", registries.Mergers.RegisteredArtifactKinds(pipeline.DefaultMergeModule), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind, dnd.ItemEventListKind, dnd.EnemyEventListKind})
assertProductionContains(t, "normalizer variants", registries.Normalizers.RegisteredArtifactKinds(pipeline.DefaultNormalizeModule), []contracts.ArtifactKind{dnd.SpellListKind, dnd.NPCListKind, dnd.CombatTurnListKind, dnd.ItemEventListKind, dnd.EnemyEventListKind})
assertProductionContains(t, "spell normalizer variants", registries.Normalizers.RegisteredArtifactKinds(spellnormalize.Key), []contracts.ArtifactKind{dnd.SpellListKind})
assertProductionContains(t, "combat normalizer variants", registries.Normalizers.RegisteredArtifactKinds(combatnormalize.Key), []contracts.ArtifactKind{dnd.CombatTurnListKind})
assertProductionContains(t, "item event normalizer variants", registries.Normalizers.RegisteredArtifactKinds(itemeventnormalize.Key), []contracts.ArtifactKind{dnd.ItemEventListKind})
assertProductionContains(t, "enemy event normalizer variants", registries.Normalizers.RegisteredArtifactKinds(enemyeventnormalize.Key), []contracts.ArtifactKind{dnd.EnemyEventListKind})
wantChain := []pipeline.ModuleBinding{
pipeline.Binding("generic/valid_json"),
@@ -132,6 +141,17 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
if got := registries.ValidatorChains.Validators(pipeline.StageNormalize, itemeventnormalize.Key); !reflect.DeepEqual(got, itemEventNormalizeChain) {
t.Fatalf("item event normalize validator chain = %#v, want %#v", got, itemEventNormalizeChain)
}
enemyEventExtractChain := []pipeline.ModuleBinding{
pipeline.Binding("generic/valid_json"),
pipeline.Binding("extract/dnd/enemy-events/shape"),
pipeline.Binding("extract/dnd/enemy-events/engagements"),
pipeline.Binding("extract/dnd/enemy-events/source_refs"),
pipeline.Binding("generic/valid_json_schema"),
pipeline.Binding("extract/dnd/enemy-events/source_relatedness"),
}
if got := registries.ValidatorChains.Validators(pipeline.StageExtract, enemyeventextract.Key); !reflect.DeepEqual(got, enemyEventExtractChain) {
t.Fatalf("enemy event extract validator chain = %#v, want %#v", got, enemyEventExtractChain)
}
assetNames := productionAssetNames(t, components.assets.PromptFS)
requiredAssets := []string{
@@ -162,6 +182,10 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
"dnd.item_events/sharedassets/common-dnd-system.md",
"dnd.item_events/sharedassets/common-dnd-transcript.md",
"dnd.item_events/task.md",
"dnd.enemy_events/dnd.enemy_events.yaml",
"dnd.enemy_events/grounding.md",
"dnd.enemy_events/instructions.md",
"dnd.enemy_events/task.md",
}
assertProductionContains(t, "production prompt assets", assetNames, requiredAssets)
@@ -180,6 +204,7 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
{stage: pipeline.StageExtract, key: "dnd/item-events", want: contracts.ExecutionClassLLMBacked},
{stage: pipeline.StageExtract, key: "dnd/npc-interactions", want: contracts.ExecutionClassLLMBacked},
{stage: pipeline.StageExtract, key: "dnd/scene-descriptions", want: contracts.ExecutionClassLLMBacked},
{stage: pipeline.StageExtract, key: enemyeventextract.Key, want: contracts.ExecutionClassLLMBacked},
{stage: pipeline.StageMerge, key: "appendorder", want: contracts.ExecutionClassDeterministic},
{stage: pipeline.StageNormalize, key: "noop", want: contracts.ExecutionClassDeterministic},
{stage: pipeline.StageNormalize, key: "dnd/spells", want: contracts.ExecutionClassDeterministic},
@@ -188,6 +213,7 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
{stage: pipeline.StageNormalize, key: "dnd/item-events", want: contracts.ExecutionClassDeterministic},
{stage: pipeline.StageNormalize, key: "dnd/npc-interactions", want: contracts.ExecutionClassDeterministic},
{stage: pipeline.StageNormalize, key: "dnd/scene-descriptions", want: contracts.ExecutionClassDeterministic},
{stage: pipeline.StageNormalize, key: enemyeventnormalize.Key, want: contracts.ExecutionClassDeterministic},
{stage: pipeline.StageOutput, key: "json", want: contracts.ExecutionClassDeterministic},
} {
got, ok := catalog.ExecutionClass(test.stage, test.key)
@@ -211,6 +237,10 @@ func TestProductionCatalogCoversMaintainedConfigurations(t *testing.T) {
if !ok || itemEventCodecSpec.Kind != dnd.ItemEventListKind || itemEventCodecSpec.Schema.ID != itemeventcodec.SchemaID {
t.Fatalf("item event codec spec = %#v, ok=%t, want typed D&D item-event codec", itemEventCodecSpec, ok)
}
enemyEventCodecSpec, ok := catalog.ArtifactCodecs.Spec(dnd.EnemyEventListKind)
if !ok || enemyEventCodecSpec.Kind != dnd.EnemyEventListKind || enemyEventCodecSpec.Schema.ID != enemyeventcodec.SchemaID {
t.Fatalf("enemy event codec spec = %#v, ok=%t, want typed D&D enemy-event codec", enemyEventCodecSpec, ok)
}
if got := catalog.ValidatorChains.Validators(pipeline.StageExtract, spells.Key); !reflect.DeepEqual(got, wantChain) {
t.Fatalf("catalog validator chain = %#v, want %#v", got, wantChain)
}

View File

@@ -153,10 +153,10 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
resume := fs.Bool("resume", false, "reuse compatible recorded checkpoints")
recomputeStep := singleValueFlag{name: "--recompute-step"}
chunkCache := chunkCacheFlag{}
sessionID := sessionIDFlag{}
requestedSessionID := sessionIDFlag{}
referenceFlags := stringListFlag{}
withoutReferenceFlags := stringListFlag{}
fs.Var(&sessionID, "session-id", "prompt session identifier")
fs.Var(&requestedSessionID, "session-id", "prompt session identifier")
fs.Var(&reasoningEffort, "reasoning-effort", "reasoning effort override")
fs.Var(&chunkCache, "chunk_cache", "chunk plan cache mode: auto, bypass, or refresh")
fs.Var(&referenceFlags, "reference", "reference binding, as slot=path, chunk.slot=path, merge.slot=path, lane.slot=path, lane.extract.slot=path, lane.merge.slot=path, or lane.normalize.slot=path")
@@ -199,7 +199,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
fmt.Fprintln(stderr, "notarius: --debug-dir must not be empty")
return 2
}
if sessionID.set && strings.TrimSpace(sessionID.value) == "" {
if requestedSessionID.set && strings.TrimSpace(requestedSessionID.value) == "" {
fmt.Fprintln(stderr, "notarius: --session-id must not be empty")
return 2
}
@@ -414,11 +414,19 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
if err != nil {
return failPipelineCommand(stderr, commandState, terminalWriter, fmt.Errorf("read input %q: %w", strings.TrimSpace(*inputPath), err))
}
effectiveSessionID, err := resolvePromptSessionID(requestedSessionID.value, effective.ResolvedPipeline.Input.Module, rawInput)
if err != nil {
return failPipelineCommand(stderr, commandState, terminalWriter, err)
}
invocation.SessionID = effectiveSessionID
if err := writeSummary(summary, func() error { return summary.WriteInvocation(invocation) }); err != nil {
return failPipelineCommand(stderr, commandState, terminalWriter, fmt.Errorf("write debug invocation metadata: %w", err))
}
chunkPlans, err := chunkPlanStoreForRun(effective.Config.Cache.ChunkPlans, opts)
if err != nil {
return failPipelineCommand(stderr, commandState, terminalWriter, err)
}
checkpointRecorder, checkpointLoader, err := checkpointHandlersForRun(effective.Config.Cache.Checkpoints, opts, effective.ResolvedPipeline, prepared.CheckpointFingerprints(), llmFingerprints, rawInput, only, llmProfiles, strings.TrimSpace(*llmProfile), strings.TrimSpace(sessionID.value), runtimeOverrides, *resume)
checkpointRecorder, checkpointLoader, err := checkpointHandlersForRun(effective.Config.Cache.Checkpoints, opts, effective.ResolvedPipeline, prepared.CheckpointFingerprints(), llmFingerprints, rawInput, only, llmProfiles, strings.TrimSpace(*llmProfile), effectiveSessionID, runtimeOverrides, *resume)
if err != nil {
return failPipelineCommand(stderr, commandState, terminalWriter, err)
}
@@ -427,7 +435,7 @@ func runPipelineCommand(args []string, stdout, stderr io.Writer, opts Options) i
Prepared: prepared,
Path: strings.TrimSpace(*inputPath),
RawInput: rawInput,
SessionID: strings.TrimSpace(sessionID.value),
SessionID: effectiveSessionID,
RunID: runID,
StartedAt: startedAt,
LLMProfiles: llmProfiles,

View File

@@ -492,19 +492,30 @@ func TestEffectiveLLMProfileIDsAreSortedDeduplicatedAndLLMOnly(t *testing.T) {
}
}
func TestRunSessionIDUsesExplicitValueOrSourceDocumentID(t *testing.T) {
func TestRunSessionIDUsesEffectiveValueForPromptRequests(t *testing.T) {
for _, tt := range []struct {
name string
args []string
want string
}{
{name: "source document", want: "source"},
{name: "derived default"},
{name: "explicit trimmed value", args: []string{"--session-id", " explicit-session "}, want: "explicit-session"},
} {
t.Run(tt.name, func(t *testing.T) {
roots := newStateTestRoots(t)
want := tt.want
if want == "" {
rawInput, err := os.ReadFile(roots.input)
if err != nil {
t.Fatal(err)
}
want, err = resolvePromptSessionID("", "test/input", rawInput)
if err != nil {
t.Fatal(err)
}
}
harness := newStateTestHarness()
args := append([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass"}, tt.args...)
args := append([]string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass", "--debug"}, tt.args...)
var stdout, stderr bytes.Buffer
code := RunWithOptions(args, &stdout, &stderr, harness.options())
if code != 0 || stderr.Len() != 0 {
@@ -517,10 +528,15 @@ func TestRunSessionIDUsesExplicitValueOrSourceDocumentID(t *testing.T) {
t.Fatalf("session IDs = %#v, want all prompt-facing module requests", sessions)
}
for _, session := range sessions {
if session != tt.want {
t.Fatalf("session IDs = %#v, want %q", sessions, tt.want)
if session != want {
t.Fatalf("session IDs = %#v, want %q", sessions, want)
}
}
var manifest artifacts.RunManifest
readStateTestSummaryJSON(t, onlyChildDir(t, roots.debug), "run-manifest.json", &manifest)
if session, ok := manifest.Metadata["session_id"]; !ok || session != want {
t.Fatalf("manifest session = %#v, want %q; metadata = %#v", session, want, manifest.Metadata)
}
})
}
}

26
internal/cli/session.go Normal file
View File

@@ -0,0 +1,26 @@
package cli
import (
"crypto/sha256"
"encoding/hex"
"fmt"
"strings"
)
const generatedSessionIDPrefix = "notarius:v1:"
func resolvePromptSessionID(explicitSessionID, inputModule string, rawInput []byte) (string, error) {
inputModule = strings.TrimSpace(inputModule)
if inputModule == "" {
return "", fmt.Errorf("resolve prompt session: input module key must not be empty")
}
if sessionID := strings.TrimSpace(explicitSessionID); sessionID != "" {
return sessionID, nil
}
hasher := sha256.New()
_, _ = hasher.Write([]byte(inputModule))
_, _ = hasher.Write([]byte{0})
_, _ = hasher.Write(rawInput)
return generatedSessionIDPrefix + hex.EncodeToString(hasher.Sum(nil)), nil
}

View File

@@ -0,0 +1,61 @@
package cli
import "testing"
func TestResolvePromptSessionIDUsesVersionedInputIdentity(t *testing.T) {
got, err := resolvePromptSessionID("", " seriatim/input/transcript ", []byte("{\"entries\":[\"one\"]}\n"))
if err != nil {
t.Fatal(err)
}
const want = "notarius:v1:e15fefdca48653e73248b4157900547be1fd34250c0138f8d3d1f2e89b43bb25"
if got != want {
t.Fatalf("resolved session = %q, want %q", got, want)
}
}
func TestResolvePromptSessionIDStabilityAndOverride(t *testing.T) {
rawInput := []byte("same input")
baseline, err := resolvePromptSessionID("", "input/transcript", rawInput)
if err != nil {
t.Fatal(err)
}
repeated, err := resolvePromptSessionID("", "input/transcript", rawInput)
if err != nil {
t.Fatal(err)
}
if baseline != repeated {
t.Fatalf("resolved sessions = %q and %q, want stable value", baseline, repeated)
}
differentModule, err := resolvePromptSessionID("", "input/other", rawInput)
if err != nil {
t.Fatal(err)
}
if baseline == differentModule {
t.Fatalf("resolved sessions = %q for distinct input modules", baseline)
}
differentInput, err := resolvePromptSessionID("", "input/transcript", []byte("same inpuu"))
if err != nil {
t.Fatal(err)
}
if baseline == differentInput {
t.Fatalf("resolved sessions = %q for distinct input bytes", baseline)
}
override, err := resolvePromptSessionID(" explicit-session ", "input/other", []byte("different input"))
if err != nil {
t.Fatal(err)
}
if override != "explicit-session" {
t.Fatalf("resolved override = %q, want %q", override, "explicit-session")
}
}
func TestResolvePromptSessionIDRejectsEmptyInputModule(t *testing.T) {
for _, explicitSessionID := range []string{"", "explicit-session"} {
if _, err := resolvePromptSessionID(explicitSessionID, " \t", []byte("input")); err == nil {
t.Fatalf("resolvePromptSessionID(%q) error = nil, want empty module failure", explicitSessionID)
}
}
}

View File

@@ -329,11 +329,25 @@ func TestRunUsesOneInjectedIdentityForDebugOutputAndManifest(t *testing.T) {
if manifest.StartedAt == nil || !manifest.StartedAt.Equal(wantStartedAt) {
t.Fatalf("manifest started at = %v, want %v", manifest.StartedAt, wantStartedAt)
}
rawInput, err := os.ReadFile(roots.input)
if err != nil {
t.Fatal(err)
}
wantSessionID, err := resolvePromptSessionID("", "test/input", rawInput)
if err != nil {
t.Fatal(err)
}
if sessionID, ok := manifest.Metadata["session_id"]; !ok || sessionID != wantSessionID {
t.Fatalf("manifest session = %#v, want %q; metadata = %#v", sessionID, wantSessionID, manifest.Metadata)
}
var invocation debugbundle.Invocation
readStateTestSummaryJSON(t, debugPath, "invocation.json", &invocation)
if invocation.RunID != runID || !invocation.StartedAt.Equal(wantStartedAt) {
t.Fatalf("debug invocation identity = %#v, want run %q at %v", invocation, runID, wantStartedAt)
}
if invocation.SessionID != wantSessionID {
t.Fatalf("debug invocation session = %q, want %q", invocation.SessionID, wantSessionID)
}
report := readStateTestRunReport(t, debugPath)
if !report.Succeeded || report.RunID != runID || report.PipelineID != "sample" || report.OutputPath != outputPath || report.DebugPath != debugPath || report.OutputCount != 1 || report.RejectedCount != 0 || report.WarningCount != 0 || report.ValidationStatus != "approved" {
t.Fatalf("success report = %#v", report)
@@ -343,6 +357,45 @@ func TestRunUsesOneInjectedIdentityForDebugOutputAndManifest(t *testing.T) {
}
}
func TestRunCheckpointReuseRequiresSameEffectiveSession(t *testing.T) {
roots := newStateTestRoots(t)
harness := newStateTestHarness()
run := func(extra ...string) stateTestResult {
args := []string{"run", "sample", "--config", roots.config, "--input", roots.input, "--chunk_cache", "bypass"}
args = append(args, extra...)
var stdout, stderr bytes.Buffer
return stateTestResult{code: RunWithOptions(args, &stdout, &stderr, harness.options()), stdout: stdout.String(), stderr: stderr.String()}
}
if result := run(); result.code != 0 {
t.Fatalf("initial run code=%d stdout=%q stderr=%q", result.code, result.stdout, result.stderr)
}
harness.mu.Lock()
initialExtractCalls := harness.extractCalls
harness.mu.Unlock()
if initialExtractCalls != 1 {
t.Fatalf("initial extract calls = %d, want 1", initialExtractCalls)
}
if result := run("--resume"); result.code != 0 {
t.Fatalf("same-session resume code=%d stdout=%q stderr=%q", result.code, result.stdout, result.stderr)
}
harness.mu.Lock()
reusedExtractCalls := harness.extractCalls
harness.mu.Unlock()
if reusedExtractCalls != initialExtractCalls {
t.Fatalf("extract calls after same-session resume = %d, want %d", reusedExtractCalls, initialExtractCalls)
}
if result := run("--resume", "--session-id", "different-session"); result.code != 0 {
t.Fatalf("different-session resume code=%d stdout=%q stderr=%q", result.code, result.stdout, result.stderr)
}
harness.mu.Lock()
differentSessionExtractCalls := harness.extractCalls
harness.mu.Unlock()
if differentSessionExtractCalls != initialExtractCalls+1 {
t.Fatalf("extract calls after different-session resume = %d, want %d", differentSessionExtractCalls, initialExtractCalls+1)
}
}
func TestRunWritesTerminalArtifactsForResolutionPipelineAndOutputFailures(t *testing.T) {
for _, tc := range []struct {
name string

View File

@@ -34,6 +34,8 @@ type ConcurrencyConfig struct {
defaultedExtractWorkers int
}
const defaultLLMConcurrency = 16
type OutputConfig struct {
Directory string `json:"directory"`
}
@@ -60,9 +62,9 @@ func Default() Config {
return Config{
Pipelines: map[string]pipeline.PipelineProfile{},
Concurrency: ConcurrencyConfig{
TotalLLM: 1,
StageWorkers: map[string]int{"extract": 1},
defaultedExtractWorkers: 1,
TotalLLM: defaultLLMConcurrency,
StageWorkers: map[string]int{"extract": defaultLLMConcurrency},
defaultedExtractWorkers: defaultLLMConcurrency,
},
Output: OutputConfig{Directory: "./notarius-output"},
Cache: CacheConfig{ChunkPlans: ChunkPlanCacheConfig{Mode: pipeline.ChunkCacheAuto}},

View File

@@ -14,7 +14,7 @@ import (
func TestDefaultReturnsDocumentedValuesAndIndependentMaps(t *testing.T) {
first := Default()
if first.Concurrency.TotalLLM != 1 || first.Concurrency.StageWorkers["extract"] != 1 {
if first.Concurrency.TotalLLM != 16 || first.Concurrency.StageWorkers["extract"] != 16 {
t.Fatalf("concurrency defaults = %#v", first.Concurrency)
}
if first.Output.Directory != "./notarius-output" || first.Debug.Directory != "./notarius-debug" {
@@ -31,7 +31,7 @@ func TestDefaultReturnsDocumentedValuesAndIndependentMaps(t *testing.T) {
first.Concurrency.StageWorkers["other"] = 100
first.Pipelines["changed"] = pipeline.PipelineProfile{}
second := Default()
if second.Concurrency.StageWorkers["extract"] != 1 || len(second.Concurrency.StageWorkers) != 1 || len(second.Pipelines) != 0 {
if second.Concurrency.StageWorkers["extract"] != 16 || len(second.Concurrency.StageWorkers) != 1 || len(second.Pipelines) != 0 {
t.Fatalf("Default() returned state shared with an earlier result: %#v", second)
}
}
@@ -45,7 +45,7 @@ func TestFileConfigMinimalVersion4AppliesOverDefaults(t *testing.T) {
if cfg.Output.Directory != "./notarius-output" || cfg.Debug.Directory != "./notarius-debug" || cfg.Cache.ChunkPlans.Mode != pipeline.ChunkCacheAuto {
t.Fatalf("minimal file changed unrelated defaults: %#v", cfg)
}
if cfg.Concurrency.TotalLLM != 1 || cfg.Concurrency.StageWorkers["extract"] != 1 || len(cfg.Pipelines) != 0 {
if cfg.Concurrency.TotalLLM != 16 || cfg.Concurrency.StageWorkers["extract"] != 16 || len(cfg.Pipelines) != 0 {
t.Fatalf("minimal file did not retain defaults: %#v", cfg)
}
}

View File

@@ -179,6 +179,30 @@ func TestWriteInvocationPreservesReasoningEffortOverrideStates(t *testing.T) {
})
}
}
func TestWriteInvocationOmitsEmptySessionID(t *testing.T) {
for _, sessionID := range []string{"", "session-123"} {
bundle, err := Allocate(t.TempDir(), testBundleRunID, time.Unix(0, 42))
if err != nil {
t.Fatal(err)
}
if err := bundle.Summary().WriteInvocation(Invocation{Operation: "run", SessionID: sessionID}); err != nil {
t.Fatal(err)
}
data, err := os.ReadFile(filepath.Join(bundle.SummaryRoot(), ArtifactInvocationMetadata))
if err != nil {
t.Fatal(err)
}
var payload map[string]any
if err := json.Unmarshal(data, &payload); err != nil {
t.Fatal(err)
}
value, found := payload["session_id"]
if found != (sessionID != "") || (found && value != sessionID) {
t.Fatalf("session found=%t value=%#v, want found=%t value=%q; JSON=%s", found, value, sessionID != "", sessionID, data)
}
}
}
func TestSummaryWriterInternalWritesConfineArtifacts(t *testing.T) {
bundle, err := Allocate(t.TempDir(), testBundleRunID, time.Unix(0, 42))
if err != nil {

View File

@@ -40,6 +40,7 @@ type Invocation struct {
OnlyLanes []string `json:"only_lanes,omitempty"`
ChunkCacheOverride string `json:"chunk_cache_override,omitempty"`
ReasoningEffortOverride *string `json:"reasoning_effort_override,omitempty"`
SessionID string `json:"session_id,omitempty"`
RunID string `json:"run_id"`
StartedAt time.Time `json:"started_at"`
}

Some files were not shown because too many files have changed in this diff Show More