106 Commits

Author SHA1 Message Date
74e2d21de5 Close the completed roadmap documents 2026-08-10 03:24:49 +00:00
7cb18a1a40 Reconcile promotion and manifest documentation 2026-08-10 03:04:20 +00:00
b556fc2f4f Clear superseded session stage result details 2026-08-10 02:53:28 +00:00
b99bd38eb4 Harden bundle promotion against symlink replacement 2026-08-10 02:45:51 +00:00
701b6726d7 Reconcile Notarius extraction documentation 2026-08-10 02:10:37 +00:00
665039f4dc Support atomic directory promotion across platforms 2026-08-10 01:58:52 +00:00
ef8dae776e Enforce canonical Notarius bundle paths 2026-08-10 01:50:31 +00:00
d01775b68a Exclude staged Notarius bundles from publish uploads 2026-08-10 01:43:43 +00:00
0d6f2dd0ce Invalidate downstream results when stages are replaced 2026-08-10 01:38:24 +00:00
df40cbec6e Document and validate Notarius extraction workflows 2026-08-10 00:42:44 +00:00
0341e0c7c0 Publish and inspect configured extraction artifacts 2026-08-10 00:32:24 +00:00
39af7d4f3c Integrate extraction artifacts into analysis catalog 2026-08-10 00:24:02 +00:00
bba582b4ca Integrate extraction lifecycle and resume validation 2026-08-10 00:14:46 +00:00
1f16a85330 Implement direct Notarius extraction execution 2026-08-09 23:59:50 +00:00
f9482639d4 Add the Notarius subprocess adapter 2026-08-09 23:47:38 +00:00
dce721cdbd Add safe immutable directory promotion 2026-08-09 23:37:02 +00:00
98734644d6 Add Notarius configuration and extraction source policy 2026-08-09 23:30:32 +00:00
951383226c Add artifact provenance and stage skip outcomes 2026-08-09 23:20:32 +00:00
df58595d1e Remove completed documentation alignment roadmaps 2026-08-09 22:02:07 +00:00
c3c14e7468 Complete documentation alignment roadmap 2026-08-09 21:53:04 +00:00
e7319ea016 Align internal documentation and maintained examples 2026-08-09 21:50:21 +00:00
bd2d5e2496 Clarify user and integration documentation contracts 2026-08-09 21:41:23 +00:00
115a44f629 Align documentation entry points and internal overview 2026-08-09 21:32:11 +00:00
18411dc5b5 Move contributor guidance to its canonical location 2026-08-09 21:27:58 +00:00
e23dc1ab6e Refocus the Narratio architecture policy 2026-08-09 21:26:18 +00:00
e1359ea227 Adopt canonical documentation ownership policy 2026-08-09 21:23:50 +00:00
7fdd99ec27 Prepare roadmap for documentation policy update 2026-08-09 21:21:00 +00:00
a90231ce0c Implement support for passing a session_id variable to scriptorium to support sticky routing 2026-07-02 21:04:37 -05:00
ed879b8bb0 Clean up obsolete placeholder code 2026-07-02 20:47:08 -05:00
717451512a Implemented new campaign/session stable inputs and corresponding input source references
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-27 09:34:05 -05:00
3ddb3a947b Update pipeline defaults so trim is enabled when omitted 2026-05-27 08:35:02 -05:00
c6632d5576 Bugfix in the seriatim adapter
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-27 08:09:22 -05:00
ffc07922c7 Cleanup following the render stage implementation and remove the completed roadmap
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-25 08:35:18 -05:00
f3310d4d16 Finalize render documentation across operations, integrations, troubleshooting, and roadmap status 2026-05-25 00:48:15 +00:00
88cee96d8d Finish render rollout with markdown publish defaults, analyze guidance, and docs updates 2026-05-25 00:46:27 +00:00
2fece10215 Implement render stage runtime and integrate it into pipeline execution 2026-05-25 00:40:06 +00:00
0658f2f642 Add render artifact model, config, and Seriatim adapter contracts 2026-05-25 00:28:01 +00:00
a51228c803 Add a documentation roadmap for the upcoming render stage feature 2026-05-24 19:15:42 -05:00
4491fb5ccd Final documentation cleanup for v1.0.0 release
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-23 11:28:03 -05:00
30b905765c Remove the deprecated narratio resume command 2026-05-23 11:25:08 -05:00
03eac70881 Mark cleanup roadmap stages as implemented 2026-05-23 16:05:21 +00:00
0f7e6b979f Deduplicate locks add/remove session-id and source parsing 2026-05-23 16:03:20 +00:00
c366912586 Extract shared read-only session inspection checks 2026-05-23 16:00:31 +00:00
9fe44cd00d Centralize Scriptorium input source policy across config, analyze, and previous-cache 2026-05-23 15:50:29 +00:00
094b0d2532 Centralize path-safe root joins and atomic file operations 2026-05-23 15:45:31 +00:00
98649f4d81 Add a roadmap to implement the remaining items identfied by the code quality audit 2026-05-23 10:35:26 -05:00
8a559efd5b Audit code quality and deduplication opportunities 2026-05-23 10:10:21 -05:00
72deccb4e2 Implement final changes from the code quality and deduplication opportunity audit 2026-05-23 10:05:14 -05:00
5620fc5bcf Refresh CLI and internal restore documentation for current behavior 2026-05-23 14:07:11 +00:00
be57e675e0 Split operator helper implementations by command responsibility 2026-05-23 14:03:31 +00:00
3971443831 Centralize remote current-state loading and preserve caller policy 2026-05-23 13:56:53 +00:00
a6b0c33e9f Unify session-aware CLI parsing and add session-id compatibility 2026-05-23 13:47:39 +00:00
96b886e711 Align internal publish terminology across stage, app, and artifacts 2026-05-23 13:39:15 +00:00
7d584ee6cd Centralize artifact source and publish destination policy 2026-05-23 13:28:25 +00:00
572a112c31 Consolidate path safety, temp downloads, and cleanup validation helpers 2026-05-23 13:20:28 +00:00
ea87c335d6 Add a roadmap to implement the high-priority items revealed by the code quality audit 2026-05-23 08:10:00 -05:00
7169ff04df Audit code quality and deduplication opportunities 2026-05-23 08:08:24 -05:00
ef1f650bc0 Mark documentation roadmap complete after final validation sweep 2026-05-23 13:04:16 +00:00
0d02cb9fa0 Rewrite integration documentation and verify maintained examples 2026-05-23 13:01:17 +00:00
0299b128cf Rewrite internal documentation for current stage and state contracts 2026-05-23 12:57:59 +00:00
d723384888 Rewrite user and operator documentation for current CLI and config behavior 2026-05-23 12:50:46 +00:00
54228055c8 Audit Stage 1 documentation scope and fix broken references 2026-05-23 12:43:50 +00:00
23ed716450 Added a documentation update roadmap 2026-05-23 07:38:17 -05:00
ab59bab044 Initial documentation cleanup pass 2026-05-23 07:11:17 -05:00
71395bb076 Rewrite docs for the publish stage contract and current behavior 2026-05-23 04:51:16 +00:00
79737edf79 Rename publish runtime terminology to published outputs 2026-05-23 04:42:08 +00:00
df2c765b7f Rename archive config and stage contract to publish 2026-05-23 04:30:45 +00:00
f050b9dd54 Added roadmap documentation for the upcoming refactoring of the publish stage 2026-05-22 23:19:44 -05:00
9c9cb54339 Implemented multiple campaign support via a campaign directory registry with explicit campaign IDs 2026-05-22 23:01:27 -05:00
7657ec3ad6 CLI cleanup to consolidate session-related subcommands 2026-05-22 22:09:17 -05:00
cee52aa092 Updated transcript artifact names and canonical paths to use a consistent, role-based nomenclature 2026-05-22 19:05:23 -05:00
e920f3a8d5 Cleaned up and removed legacy configuration surfaces 2026-05-22 18:32:14 -05:00
591c529a09 Updated the analyze stage to accept --artifacts as a CLI flag 2026-05-22 18:01:05 -05:00
7324c5a686 Session configuration templates are now proceeded by narratio session init; all other commands require concrete configuration 2026-05-22 17:38:23 -05:00
d0936fb022 Implemented default config/campaign discovery for narratio session init 2026-05-22 11:36:57 -05:00
2aa074c5cf Implemented narratio publish as a shortcut to run the archive stage only 2026-05-22 11:28:38 -05:00
782d0cf3b9 Upgraded the restore command to download previous session artifcats when configured as inputs for the current session analyze stage 2026-05-21 23:31:01 -05:00
083c01cfa0 Implemented narratio analyze as a shortcut to run the analyze stage only
Some checks failed
ci/woodpecker/tag/release Pipeline failed
2026-05-21 23:02:19 -05:00
2937696024 Add clean command 2026-05-21 22:48:18 -05:00
b817a5b772 Implemented shared S3 audio caching for prepare and restore --include-audio 2026-05-21 22:22:08 -05:00
3022f20beb Simplified the output of narratio artifacts list --remote and narratio status --session-id 2026-05-21 21:20:41 -05:00
ca1ded1821 Bugfix for commands that list artifacts in the S3 backend 2026-05-21 21:00:51 -05:00
3752f3ed28 Added remote artifact listing to narratio status 2026-05-21 20:49:28 -05:00
870c2d69d5 Consolidated addition, removal, and listing of locks under a single narratio locks command 2026-05-21 19:36:14 -05:00
135407ba7c Implemented a centralized secret-backed object-store helper 2026-05-21 19:13:10 -05:00
228c348e42 Implemented operations helper commands for validation, locking, and status 2026-05-21 11:50:20 -05:00
a813bd5a50 Fixed a redundant path bug for previous session artifacts 2026-05-21 10:58:49 -05:00
d8f58dce31 Normalize the default configuration discovery paths for all three config files, and update documentation and tests accordingly 2026-05-21 09:55:56 -05:00
7111edeca4 Add archive promotion locks 2026-05-20 21:40:09 -05:00
3aae4bbb12 Add remote session loading 2026-05-20 20:55:13 -05:00
b29d8eeb50 Add campaign configuration support 2026-05-20 20:41:28 -05:00
dffb432537 Removed completed roadmap for previous_session artifacts 2026-05-20 20:15:47 -05:00
2dd38c7913 Refine campaign and remote session roadmap 2026-05-20 20:15:05 -05:00
bc2ade38d9 Finalize previous-session artifact documentation and restore-analyze continuity coverage 2026-05-20 15:17:04 +00:00
5be831eb13 Restore archived previous-session cache files with session state 2026-05-20 15:05:48 +00:00
cae4d99a89 Archive durable previous-session cache files with session state 2026-05-20 15:03:30 +00:00
e09dc0512d Add analyze integration coverage for previous-session inputs 2026-05-20 15:01:27 +00:00
ae82bc1ce0 Resolve canonical previous-session artifact sources from prepared previous cache 2026-05-20 14:59:28 +00:00
01eb7aa1aa Add prepare rerun guidance for unresolved previous-session analyze inputs 2026-05-20 14:55:44 +00:00
2ca700195c Integrate previous-session artifact hydration into prepare stage 2026-05-20 14:53:22 +00:00
2b08c34539 Add prepare helper to hydrate previous-session artifacts from archive 2026-05-20 14:49:22 +00:00
79f1fc1e09 Add helper to collect previous-session artifact input requirements 2026-05-20 14:37:02 +00:00
9c753270bd Add canonical previous-session artifact source parsing and validation 2026-05-20 14:34:23 +00:00
b907cb01aa Add previous-session workspace path helpers and layout support 2026-05-20 14:31:03 +00:00
7824afd4a5 Add previous session ID templating and CLI support 2026-05-20 14:26:32 +00:00
2a4e1e912c Update documentation to include a roadmap for previous session artifact support 2026-05-20 09:12:17 -05:00
237 changed files with 27344 additions and 6815 deletions

BIN
.DS_Store vendored

Binary file not shown.

View File

@@ -1,22 +1,42 @@
# narratio
Narratio is a Go orchestration application that turns D&D session audio into polished transcripts and generated session artifacts.
Narratio is a stage-driven Go orchestrator for turning D&D session audio into
polished transcripts, validated Notarius extraction lanes, and generated
artifacts.
It coordinates transcription, merge/polish/normalize/trim processing, artifact generation, archive publishing, and resumable run state in one operator workflow.
It runs a deterministic workflow with manifest-driven continuation, remote
publish, and restore support.
```bash
narratio run --session-id 2026-04-04
```sh
narratio run 2026-04-04
```
This command requires discoverable `pipeline.yml` and `session.yml` files (or explicit `--config` and `--session` flags).
This requires resolvable `pipeline.yml`, `campaign.yml`, and concrete
`session.yml` files or their explicit command-line alternatives.
## Documentation
- [Configuration](docs/config.md)
- [CLI Reference](docs/cli.md)
- [Operations and Recovery](docs/operations.md)
- [Troubleshooting](docs/troubleshooting.md)
- [Development Guide](docs/development.md)
- [Architecture Principles](docs/architecture.md)
- [Internal Component Contracts](docs/internal/README.md)
- [Config Examples](examples/)
- [CLI reference](docs/cli.md) — commands, arguments, flags, and invocation
behavior.
- [Configuration](docs/config.md) — discovery, fields, defaults, and
validation.
- [Operations](docs/operations.md) — runtime workflow, state, publishing,
recovery, and cleanup.
- [Troubleshooting](docs/troubleshooting.md) — symptom-driven diagnosis and
safe remedies.
- [Integration contracts](docs/integrations/) — external tools, formats, and
compatibility expectations.
- [Maintained examples](examples/README.md) — complete copyable configuration
and input files.
## Maintainer Documentation
- [Development guide](docs/development.md) — first-read orientation and
task-specific reading routes.
- [Internal overview](docs/internal/overview.md) — implemented component map.
- [Architecture](docs/policy/architecture.md) — normative boundaries and
invariants.
- [Documentation policy](docs/policy/documentation.md) — canonical ownership
and maintenance rules.
- [Testing policy](docs/policy/testing.md) — test value, boundaries, and
sufficiency.

View File

@@ -1,202 +0,0 @@
# Narratio Architecture
## Purpose
`narratio` is a Go orchestration application for processing D&D session audio into polished transcripts and generated session artifacts.
This document defines the development principles for the project. It is inward-facing: its audience is developers and LLM coding agents. It should guide future changes, not serve as a complete implementation reference.
Implemented component details belong under `docs/internal/`.
## Project Shape
Narratio is a modular, stage-driven orchestrator.
It coordinates specialized downstream systems rather than reimplementing their domains:
- WhisperX handles transcription.
- Seriatim handles deterministic transcript merge/normalization/trim behavior.
- Audita handles transcript correction and polishing.
- Scriptorium handles prompt execution and generated artifacts.
Narratio owns orchestration, configuration loading, session/run state, local and remote path modeling, manifest persistence, stage sequencing, resume behavior, and archive semantics.
Narratio should remain explicit and comprehensible. It is not intended to become a generic workflow engine.
## Core Principles
### Modular and composable
Code should be organized around clear responsibilities. Stages, adapters, config loading, manifest persistence, path construction, and storage behavior should remain separable and independently testable.
### Hexagonal boundaries
External systems should be isolated behind narrow adapters. Stage logic should depend on Narratio-level interfaces and data structures, not on external SDK types, subprocess argument construction, or transport-specific details.
### Standard library preference
Prefer the Go standard library. Add dependencies only when they provide substantial value, are necessary for an external integration, or are a widely used de facto standard.
Accepted examples include a YAML library for configuration and the AWS SDK for S3-compatible storage.
### Explicit orchestration
The pipeline should remain stage-driven and explicit. New behavior should be added through clear stage, adapter, config, or manifest contracts rather than implicit side effects or generic workflow abstraction.
## Stage Design
Each stage should have a clear scope of responsibility.
A stage should define:
- its purpose;
- required input state;
- produced output state;
- config fields it consumes;
- external adapters it uses;
- manifest refs it reads or writes;
- skip, force, and resume behavior;
- failure behavior;
- tests that protect its contract.
Stages should avoid reaching across boundaries. If shared behavior is needed, prefer a helper or service with a narrow interface over duplicating ad hoc logic between stages.
## Transactionality and Resume
A stage should behave transactionally.
A stage is complete only when its outputs have been written, validated, and recorded in the manifest. If a stage fails, Narratio should preserve enough local state for inspection, recovery, and resume.
A failed or incomplete run must not be treated as successful. Later stages should depend on manifest-recorded success, not merely on incidental files existing on disk.
## Manifest Model
The manifest is the durable local ledger for a run.
It should record:
- run identity;
- stage status;
- input and output refs;
- logs and generated config refs;
- checksums or provenance where useful;
- non-secret adapter and archive metadata.
Resume behavior should be manifest-driven. Filesystem state may be inspected and validated, but it should not replace manifest stage state as the source of run progress.
## Adapter Boundaries
Adapters own external integration details.
Expected boundaries:
- WhisperX HTTP details stay in the WhisperX adapter.
- Seriatim CLI construction stays in the Seriatim adapter.
- Audita CLI construction stays in the Audita adapter.
- Scriptorium CLI construction stays in the Scriptorium adapter.
- Object-storage details stay behind the storage adapter interface.
- AWS SDK types stay inside the S3 storage implementation.
Stage code should express intent in Narratio terms and call adapters through narrow contracts.
## Configuration Philosophy
Configuration should be strict, explicit, and operator-friendly.
Principles:
- YAML decoding should reject unknown fields.
- Defaults should be centralized and testable.
- Empty configured values should not silently override meaningful defaults.
- Session templating should remain narrow and deterministic.
- Template support should serve operator convenience, not become a general configuration language.
Narratio should not become a secondary configuration system for downstream tools. Seriatim, Audita, and Scriptorium should own their runtime defaults wherever practical. Narratio should pass required stage-contract paths and explicit operator overrides.
## Path and Storage Discipline
Local and remote paths are part of Narratios application contract.
Code should use centralized path helpers for workspace, spool, session, run, artifact, log, config, and archive paths. Stages should avoid reconstructing canonical paths through scattered string concatenation.
Storage backends should receive explicit bucket-relative keys. Storage implementations should not infer campaign, session, run, or root-prefix semantics.
## Archive Invariants
Archive behavior must preserve a clear commit boundary.
A remote run is current only after the archive stage has successfully uploaded the run record, required promoted outputs, `current/manifest.json`, and finally `current/run_id.txt`.
`current/run_id.txt` is the final remote commit marker and must be written last.
Failed, incomplete, skipped, or uncommitted archive attempts must not be presented as current remote state. Local cleanup is permitted only after successful archive commit and only when explicitly configured.
## Security and Privacy
Narratio handles private campaign material.
Rules:
- Do not store raw secrets in pipeline or session YAML.
- Use environment variable names or secret-file references for secret handling.
- Do not write raw secret values to manifests, logs, generated configs, or archive metadata.
- Treat transcripts, generated artifacts, prompts, reports, and logs as potentially sensitive.
- Avoid logging transcript or prompt content unless there is a deliberate diagnostic reason.
## Diagnostics
Diagnostics should be durable and discoverable, but distinct from canonical outputs.
Logs, reports, generated invocation/config files, and render-debug files support debugging. Transcript tiers and configured artifacts are pipeline products.
Manifest refs should preserve that distinction.
## Determinism
Where practical, Narratio should prefer deterministic behavior:
- stable local path layout;
- stable remote key layout;
- sorted upload order;
- predictable generated config files;
- repeatable command construction;
- tests that do not depend on live external services.
Run IDs and timestamps may be intentionally variable, but surrounding behavior should remain testable.
## Testing Expectations
Core behavior should be testable without live external services.
Tests should cover:
- config loading, defaults, and validation;
- CLI parsing and command construction;
- path helpers;
- manifest transitions;
- stage success, failure, skip, and resume behavior;
- adapter command construction;
- fake storage behavior;
- archive commit ordering;
- example config validity where practical.
Live S3, WhisperX, LLM, or subprocess integration tests should be explicit integration tests, not required for ordinary unit test runs.
## Documentation Expectations
Documentation must follow `docs/documentation/policy.md`.
Current behavior belongs in user-facing docs and `docs/internal/`. Future, planned, aspirational, experimental, or unimplemented work belongs only under `docs/roadmap/`.
`docs/architecture.md` should remain concise and principle-focused. It should not duplicate the full config reference, CLI reference, operations guide, or internal stage documentation.
## Non-Goals
Narratio is not:
- a generic DAG or workflow engine;
- a replacement configuration layer for Seriatim, Audita, or Scriptorium;
- a storage backend abstraction beyond the needs of this pipeline;
- a place to embed raw secrets;
- a place for stage logic to depend directly on AWS SDK types or downstream tool internals;
- a prompt-authoring system.

View File

@@ -1,61 +1,92 @@
# CLI
# CLI Reference
## Shortest Useful Command
```bash
narratio run --session-id 2026-04-04
narratio run 2026-04-04
```
This command uses default discovery for `pipeline.yml` and `session.yml`; both files must be discoverable unless you pass explicit `--config` and `--session` paths.
This runs the canonical full pipeline for session `2026-04-04`.
## Command Overview
Implemented commands:
Top-level commands:
- `run`: execute pipeline stages and persist manifest state.
- `plan`: validate config, prepare workspace layout, and print stage run/skip decisions.
- `resume`: continue from first non-succeeded stage unless forced.
- `status`: read and print stage statuses from an existing manifest.
- `run-stage`: execute exactly one stage.
- `restore`: restore durable local session state from the committed remote archive state.
- `run <session_id>`: run full stage order.
- `run-stage <stage> <session_id>`: run one stage.
- `analyze <session_id>`: force-run analyze.
- `publish <session_id>`: force-run publish.
- `clean <session_id>` or `clean --all`: remove local work/spool state.
- `session <subcommand>`: session helper commands.
Unknown commands print usage and exit non-zero.
Session subcommands:
For config semantics, see [docs/config.md](./config.md). For operator lifecycle and recovery, see [docs/operations.md](./operations.md).
- `session init <session_id>`
- `session plan <session_id>`
- `session validate <session_id>`
- `session status <session_id>`
- `session restore <session_id>`
- `session artifacts <session_id>`
- `session locks <session_id>`
- `session locks add <session_id> <source>`
- `session locks remove <session_id> <source>`
## Complete Flag Reference
## Common Config Flags
Most session-aware commands accept:
- `--config <pipeline.yml>`
- `--campaign <id>`
- `--campaign-file <campaign.yml>`
- `--session <session.yml>`
- `--session-id <session_id>`
- `--previous-session-id <session_id>`
Rules:
- `--campaign` and `--campaign-file` are mutually exclusive.
- `--session` is not used by `session init`.
- if both positional `<session_id>` and `--session-id` are provided, values must match.
- `clean --all` cannot be combined with campaign/session selectors.
## Session ID Input Rules
Session-aware commands accept one of these forms:
- positional session ID: `... <session_id>`
- compatibility flag: `... --session-id <session_id>`
When both are present, command parsing requires an exact match.
Commands with additional positionals keep their command-specific order:
- `run-stage <stage> <session_id>` or `run-stage <stage> --session-id <session_id>`
- `session locks add <session_id> <source>` or `session locks add --session-id <session_id> <source>`
- `session locks remove <session_id> <source>` or `session locks remove --session-id <session_id> <source>`
## Command Reference
### `run`
- `--config <path>`: optional explicit `pipeline.yml` path.
- `--session <path>`: optional explicit `session.yml` path.
- `--session-id <value>`: session template variable value.
- `--force`: force stage execution.
- `--artifacts <names>`: analyze artifact keys to execute (repeatable or comma-separated).
```bash
narratio run <session_id> [--force] [--artifacts <name[,name...]>] [...common config flags]
```
### `plan`
Behavior:
- `--config <path>`
- `--session <path>`
- `--session-id <value>`
- `--force`
### `resume`
- `--config <path>`
- `--session <path>`
- `--session-id <value>`
- `--force`
- `--artifacts <names>`: analyze artifact keys to execute (repeatable or comma-separated).
- evaluates full stage order;
- runs `extract` between `trim` and `render`; an omitted or disabled Notarius
configuration records an explicit `notarius_disabled` self-skip;
- skips already-succeeded stages unless `--force` is set or a stage-specific
resume check finds its durable result obsolete;
- continues interrupted or partially completed sessions by running non-succeeded stages;
- writes session and run manifests.
### `run-stage`
- `--config <path>`
- `--session <path>`
- `--session-id <value>`
- `--force`
- `--artifacts <names>`: analyze artifact keys to execute (repeatable or comma-separated).
- positional `<stage>`: required stage name.
```bash
narratio run-stage <stage> <session_id> [--force] [--artifacts <name[,name...]>] [...common config flags]
```
Valid stage names:
@@ -65,214 +96,206 @@ Valid stage names:
- `polish`
- `normalize`
- `trim`
- `extract`
- `render`
- `analyze`
- `archive`
- `publish`
- `notify`
### `restore`
Rules:
- `--config <path>`
- `--session <path>`
- `--session-id <value>`
- `--dry-run`: plan restore actions without writing local files.
- `--force`: overwrite local conflicting files with remote archive files.
- `--include-audio`: include durable archived `audio/**` files in restore scope.
- `--artifacts` is accepted only for `analyze` and `publish` stage targets.
### `status`
- `--manifest <path>`: required manifest path.
## Command Reference
### `run`
Purpose:
- Execute configured stages in canonical order.
Syntax:
### `analyze`
```bash
narratio run [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force] [--artifacts <name[,name...]>]
narratio analyze <session_id> [--artifacts <name[,name...]>] [...common config flags]
```
Success output:
- `narratio run: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
Common failure cases:
- missing default config/session paths when flags omitted.
- invalid template/rendered session mismatch.
- unknown/invalid `--artifacts` value.
- `--artifacts` with unknown configured artifact key.
### `plan`
Purpose:
- Validate config, load secrets (if configured), prepare workdir, and print stage run/skip decisions.
Syntax:
Equivalent to:
```bash
narratio plan [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force]
narratio run-stage analyze <session_id> --force [...common config flags]
```
Success output includes:
- `narratio plan: workdir prepared at <path>`
- one line per stage (`<stage>: run|skip`)
- `totals: run=<n> skip=<n>`
Common failure cases:
- same config/session discovery and validation failures as `run`.
- secrets directory read failures when `pipeline.secrets.env_dir` is configured.
### `resume`
Purpose:
- Continue from session-manifest stage status.
Syntax:
### `publish`
```bash
narratio resume [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force] [--artifacts <name[,name...]>]
narratio publish <session_id> [--artifacts <name[,name...]>] [...common config flags]
```
Success output:
- `narratio resume: session <session_id> has no remaining stages`
- or `narratio resume: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
Common failure cases:
- same discovery/template/validation failures as `run`.
- manifest load errors when existing manifest is unreadable.
- invalid or unknown artifact selections.
### `status`
Purpose:
- Inspect one manifest file without executing stages.
Syntax:
Equivalent to:
```bash
narratio status --manifest <manifest.json>
narratio run-stage publish <session_id> --force [...common config flags]
```
Success output includes:
- `session_id: <id>`
- `updated_at: <timestamp>`
- `stages:` entries (`- <stage>: <status>`)
Common failure cases:
- missing `--manifest`.
- unreadable or invalid manifest path.
### `run-stage`
Purpose:
- Execute exactly one stage.
Syntax:
### `clean`
```bash
narratio run-stage [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force] [--artifacts <name[,name...]>] <stage>
narratio clean <session_id> [--dry-run] [--clear-cache] [...common config flags]
narratio clean --all [--dry-run] [--clear-cache] [--config <pipeline.yml>]
```
Success output:
- `narratio run-stage: stage=<name> executed=<n> skipped=<n> force=<true|false>; manifest=<path>`
Behavior:
`--artifacts` behavior:
- accepted only when `<stage>` is `analyze`.
- names are normalized (trimmed, deduplicated, sorted).
- unknown configured artifact keys fail.
- session mode removes the selected session's local work and spool state;
- `--all` removes all local session work and spool state;
- cache remains unless `--clear-cache` is provided.
Common failure cases:
- missing stage positional arg.
- unknown stage name.
- using `--artifacts` with any non-`analyze` stage.
See [Operations: Cleanup](./operations.md#cleanup) for deletion scope and
post-publish cleanup behavior.
### `restore`
Purpose:
- Restore durable session state (`manifest.json`, `transcripts/**`, `artifacts/**`, and optional `audio/**`) from the committed remote archive current state.
Syntax:
### `session plan`
```bash
narratio restore [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--dry-run] [--force] [--include-audio]
narratio session plan <session_id> [--force] [...common config flags]
```
Success output (dry-run):
- `Restore plan for <campaign>/<session_id>`
- `Remote run: <run_id>`
- `Would download: <n>`
- `Would skip unchanged: <n>`
- `Conflicts: <n>`
Validates config, prepares local workdir layout, and prints run/skip decisions for each stage.
Success output (non-dry-run):
- `Restored session archive for <campaign>/<session_id>`
- `Remote run: <run_id>`
- `Downloaded: <n>`
- `Skipped unchanged: <n>`
- `Conflicts: <n>`
### `session validate`
Common failure cases:
- storage backend is not configured.
- remote `current/run_id.txt` missing/empty.
- remote `current/manifest.json` missing or invalid.
- remote manifest session/campaign mismatch.
- local conflicts without `--force`.
- session lock conflict.
```bash
narratio session validate <session_id> [...common config flags]
```
Read-only preflight checks for config validity, required inputs, audio mode, previous-session requirements, publish outputs, and effective locks.
### `session status`
```bash
narratio session status <session_id> [...common config flags]
```
Prints local manifest state and, when storage is available, remote current-state and published-output status.
### `session init`
```bash
narratio session init <session_id> --output ./session.yml [options]
narratio session init <session_id> --remote [options]
```
Required target selection:
- exactly one of:
- `--output <path>`
- `--remote`
Options:
- `--config <pipeline.yml>`
- `--campaign <id>` or `--campaign-file <campaign.yml>`
- `--previous-session-id <id>`
- `--date <YYYY-MM-DD>`
- `--title <text>`
- `--audio-dir <path>`
- `--audio-s3-prefix <prefix>`
- `--force`
Rules:
- `--audio-dir` and `--audio-s3-prefix` are mutually exclusive.
- if campaign `session_template_file` is configured, `session init` renders it.
- generated session YAML must be concrete (no unresolved `{{ ... }}` placeholders).
### `session restore`
```bash
narratio session restore <session_id> [--dry-run] [--force] [--include-audio] [...common config flags]
```
Behavior:
- discovers committed remote current state;
- plans local restores;
- writes an execution report;
- blocks conflicting overwrites unless `--force` is set.
See [Operations: Restore Workflow](./operations.md#restore-workflow) for the
default restore scope, report location, and conflict-handling workflow.
### `session artifacts`
```bash
narratio session artifacts <session_id> [--remote] [...common config flags]
```
Lists effective built-in, configured Scriptorium, and configured extraction
sources; reports planned, available, unavailable, and published state without
reading payload bodies; and includes publish rules, lock state, and optional
remote published-state availability.
### `session locks`
```bash
narratio session locks <session_id> [...common config flags]
narratio session locks add <session_id> <source> [--reason <text>] [--force] [...common config flags]
narratio session locks remove <session_id> <source> [...common config flags]
```
Behavior:
- list mode reports the effective merge of static and remote locks;
- add/remove mutate only remote locks;
- static locks from pipeline config cannot be removed by CLI commands.
See [Operations: Publish Locks](./operations.md#publish-locks) for lock storage
and precedence.
## `--artifacts` Selection Rules
- accepted on `run`, `run-stage`, `analyze`, and `publish`;
- names must exist in `pipeline.scriptorium.artifacts`;
- empty entries are invalid;
- repeated names are deduplicated.
Effects:
- filters analyze execution to selected configured artifacts;
- filters publish rules that source `narratio.artifact.<name>`;
- does not filter built-in transcript/bounds or explicitly configured
`narratio.extraction.<name>` publish sources; and
- does not select or filter Notarius lanes.
## Common Workflows
Default-discovery run:
Run full pipeline:
```bash
narratio run --session-id 2026-04-04
narratio run 2026-04-04
```
Run only selected analyze artifacts:
Dry-run restore plan:
```bash
narratio run --session-id 2026-04-04 --artifacts session_recap,player_handout
narratio session restore 2026-04-04 --dry-run
```
Resume with selected analyze artifacts:
Generate a concrete session file from template/default structure:
```bash
narratio resume --session-id 2026-04-04 --artifacts player_handout
narratio session init 2026-04-04 --output ./session.yml --date 2026-04-04 --title "Session 12"
```
Run only analyze stage with selected artifacts:
Force publish only:
```bash
narratio run-stage --session-id 2026-04-04 --artifacts player_handout analyze
narratio publish 2026-04-04
```
Preview restore actions without writes:
## Output And Exit Behavior
```bash
narratio restore --session-id 2026-04-04 --dry-run
```
- Successful commands write their result or summary to standard output and
exit with status `0`.
- Command failures and invalid invocations write an error to standard error and
exit with status `1`.
- An unknown top-level command also prints the top-level usage summary to
standard error.
- `session restore --help` prints its command-specific usage and exits with
status `0`.
Restore and then force analyze:
```bash
narratio restore --session-id 2026-04-04
narratio run-stage --session-id 2026-04-04 --force analyze
```
## Diagnostic / Recovery Commands
Inspect stage status:
```bash
narratio status --manifest <manifest.json>
```
Get manifest path from previous output:
- `run`, `resume`, and `run-stage` print `manifest=<path>` on success.
## `--artifacts` and `--force`
- `--artifacts` filters which configured artifacts are executable when analyze runs.
- `--artifacts` does not imply `--force`.
- if analyze is already `succeeded` and `--force` is not set, runner-level skip still applies.
Output is intended for operator inspection. Narratio does not currently offer
a machine-readable CLI output mode; durable machine-readable state is recorded
in manifests and reports described in [Operations](./operations.md).

View File

@@ -1,152 +1,142 @@
# Configuration
# Configuration Reference
## 1. Overview
## Purpose
Narratio loads two YAML files:
Narratio resolves three YAML documents:
- `pipeline.yml`: pipeline-level runtime configuration.
- `session.yml`: per-session metadata and input selection.
- `pipeline.yml`: pipeline/runtime settings
- `campaign.yml`: campaign identity and stable input defaults
- `session.yml`: session identity, metadata, and audio source selection
These commands load and validate both files before running:
## Discovery and Selection
- `narratio run`
- `narratio plan`
- `narratio resume`
- `narratio run-stage`
- `narratio restore`
### `pipeline.yml`
Behavior:
When `--config` is omitted, search order is:
- strict YAML decode is enabled (`KnownFields(true)`): unknown fields fail.
- session templates render before session YAML decode.
- defaults are applied for optional pipeline fields.
- validation enforces required fields, value formats, and cross-field constraints.
1. `/usr/local/etc/narratio/pipeline.yml`
2. `/etc/narratio/pipeline.yml`
## 2. Config file discovery
### `campaign.yml`
Pipeline config lookup for `run`, `plan`, `resume`, `run-stage`, and `restore`:
Selection rules:
- if `--config <path>` is provided, that path is used.
- if omitted, Narratio searches in order:
1. `/usr/local/etc/narratio/pipeline.yml`
2. `/etc/narratio/pipeline.yml`
- first existing file wins.
- if `--campaign-file` is set, use that path;
- else if `--campaign <id>` is set, use `{pipeline.campaigns.root}/{id}/campaign.yml`;
- else use `{pipeline.campaigns.root}/{pipeline.campaigns.default_campaign_id}/campaign.yml`.
## 3. Session file discovery and templating
### `session.yml`
Session config lookup for `run`, `plan`, `resume`, `run-stage`, and `restore`:
When `--session` is omitted, local search order is:
- if `--session <path>` is provided, that path is used.
- if omitted, Narratio searches in order:
1. `./session.yml`
2. `/usr/local/etc/narratio/session.yml`
3. `/etc/narratio/session.yml`
- first existing file wins.
1. `/usr/local/etc/narratio/session.yml`
2. `/etc/narratio/session.yml`
Template behavior:
If local session discovery fails and a `session_id` is known, Narratio attempts remote session loading from:
- supported placeholders:
- `{{session_id}}`
- `{{ session_id }}`
- `--session-id <value>` supplies the placeholder value.
- unresolved placeholders fail load.
- if rendered `session_id` mismatches `--session-id`, load fails.
- `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/session.yml`
## 4. Minimal pipeline config
using configured object storage.
## Validation and Merge Rules
- YAML decode is strict (`KnownFields(true)`): unknown fields fail load.
- Session files must be concrete; unresolved `{{ ... }}` placeholders fail load.
- Pipeline defaults are applied before validation.
- Campaign and session identities must agree.
- Stable files (`speakers_file`, `autocorrect_file`, `glossary_file`, `players_file`, `party_file`) resolve from session overrides when provided, otherwise from campaign defaults.
- Exactly one audio mode must be configured in session input:
- local (`audio_dir` or `audio_files`), or
- S3 (`audio_s3.prefix`).
## Minimal Working Configuration
`pipeline.yml`
```yaml
campaigns:
root: /usr/local/share/narratio/campaigns
default_campaign_id: sample-campaign
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"
transcribe_url: https://transcription.example.com/transcribe
```
Why this is sufficient:
- `whisperx.transcribe_url` is required.
- `workspace.root` defaults to `/var/lib/narratio`.
- optional sections (`seriatim`, `audita`, `archive`, `scriptorium`, `trim`, `normalize`, etc.) receive defaults or stay inactive.
## 5. Minimal session template
`campaign.yml`
```yaml
session_id: "{{ session_id }}"
campaign: sample-campaign
campaign_id: sample-campaign
inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
players_file: ./players.yml
party_file: ./party.yml
```
`session.yml` (local audio)
```yaml
session_id: 2026-05-03
inputs:
audio_dir: ./audio
speakers_file: ./examples/speakers.yml
autocorrect_file: ./examples/autocorrect.yml
glossary_file: ./examples/glossary.yml
```
Usage:
## Secrets Handling
```bash
narratio run --config /path/to/pipeline.yml --session ./session.yml --session-id 2026-05-03
```
- Do not place raw secrets in YAML.
- Use env var names in config (for example `pipeline.audita.llm_api_key_env`).
- Optionally load env files from `pipeline.secrets.env_dir`.
- Commands that need storage/auth load filesystem secrets before constructing adapters.
## 6. Production-oriented config
## Publish Configuration Summary
Publish rules live under `pipeline.publish`.
```yaml
workspace:
root: /var/lib/narratio/workspace
cleanup_after_archive: true
storage:
backend: s3
s3:
bucket: my-dnd-archive
root_prefix: dnd
region: us-east-1
access_key_id_env: OBJECT_STORAGE_KEY_ID
secret_access_key_env: OBJECT_STORAGE_KEY
spool:
root: /var/spool/narratio
delete_audio_after_archive: true
archive:
publish:
enabled: true
upload_run: true
promote_artifacts:
- source: narratio.transcript.trimmed
dest: transcripts/trimmed.json
outputs:
- source: narratio.transcript.final_trimmed
dest: transcripts/final.trimmed.json
required: true
- source: narratio.transcript.final_markdown
dest: transcripts/final.md
required: true
- source: narratio.transcript.final_trimmed_markdown
dest: transcripts/final.trimmed.md
required: true
- source: narratio.artifact.session_recap
dest: artifacts/session_recap.md
required: true
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"
scriptorium:
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
output_path: artifacts/session_recap.md
inputs:
transcript:
source: narratio.transcript.trimmed
required: true
locks:
- source: narratio.artifact.session_recap
reason: manual post-publish edits
```
Operational notes:
Rules:
- archive promotion is explicit and source-based via `archive.promote_artifacts`.
- `source` is required; `dest` is optional and derived when omitted.
- Narratio does not auto-promote all generated analyze artifacts.
- `restore` reads the same config/session inputs and restore scope is bounded by committed archive current state.
- `outputs[].source` is required.
- `outputs[].dest` may be omitted when derivable from source.
- extraction sources require an explicit `outputs[].dest` and publish only when
a rule names that source; the Notarius index and complete bundle are not
publish sources.
- `outputs[].required` defaults to `true`.
- static locks (`pipeline.publish.locks`) merge with remote locks (`{session_prefix}/locks.yml`), with static locks taking precedence on duplicates.
## 7. Full pipeline reference
## Full Schema
| Path | Type | Required | Default |
### Pipeline
| Field | Type | Required | Default / Rule |
| --- | --- | --- | --- |
| `pipeline.workspace.root` | string | No | `/var/lib/narratio` |
| `pipeline.workspace.cleanup_after_archive` | bool | No | `false` |
| `pipeline.secrets.env_dir` | string | Conditional | none |
| `pipeline.workspace.cleanup_after_publish` | bool | No | `false` |
| `pipeline.campaigns.root` | string | No | `/usr/local/share/narratio/campaigns` |
| `pipeline.campaigns.default_campaign_id` | string | No | empty |
| `pipeline.secrets.env_dir` | string | No | empty |
| `pipeline.storage.backend` | string | No | empty |
| `pipeline.storage.bucket` | string | No | empty |
| `pipeline.storage.prefix` | string | No | empty |
| `pipeline.storage.s3.bucket` | string | Conditional | empty |
| `pipeline.storage.s3.bucket` | string | Conditional | required for S3 session-audio and for publish upload when backend is `s3` |
| `pipeline.storage.s3.root_prefix` | string | No | `dnd` |
| `pipeline.storage.s3.region` | string | No | empty |
| `pipeline.storage.s3.endpoint` | string | No | empty |
@@ -154,21 +144,26 @@ Operational notes:
| `pipeline.storage.s3.access_key_id_env` | string | No | `OBJECT_STORAGE_KEY_ID` |
| `pipeline.storage.s3.secret_access_key_env` | string | No | `OBJECT_STORAGE_KEY` |
| `pipeline.spool.root` | string | No | `/var/spool/narratio` |
| `pipeline.spool.delete_audio_after_archive` | bool | No | `false` |
| `pipeline.archive.enabled` | bool | No | `true` |
| `pipeline.archive.upload_run` | bool | No | `true` |
| `pipeline.archive.promote_artifacts[]` | list | No | trimmed transcript rule |
| `pipeline.archive.promote_artifacts[].source` | string | Yes (per rule) | none |
| `pipeline.archive.promote_artifacts[].dest` | string | No | derived from source |
| `pipeline.archive.promote_artifacts[].required` | bool | No | `true` |
| `pipeline.whisperx.transcribe_url` | string | Yes | none |
| `pipeline.spool.delete_audio_after_publish` | bool | No | `false` |
| `pipeline.cache.root` | string | No | `/var/cache/narratio` |
| `pipeline.cache.s3_audio` | bool | No | `true` |
| `pipeline.publish.enabled` | bool | No | `true` |
| `pipeline.publish.upload_run` | bool | No | `true` |
| `pipeline.publish.outputs[]` | list | No | defaults to final trimmed JSON plus final and final-trimmed Markdown outputs |
| `pipeline.publish.outputs[].source` | string | Yes (per rule) | must reference built-in or configured artifact source |
| `pipeline.publish.outputs[].dest` | string | Conditional | derived if omitted and source supports derivation |
| `pipeline.publish.outputs[].required` | bool | No | `true` |
| `pipeline.publish.locks[]` | list | No | empty |
| `pipeline.publish.locks[].source` | string | Yes (per lock) | must reference supported publish source |
| `pipeline.publish.locks[].reason` | string | No | empty |
| `pipeline.whisperx.transcribe_url` | string | Yes | valid URL |
| `pipeline.whisperx.language` | string | No | `en` |
| `pipeline.whisperx.timeout` | duration string | No | `30m` |
| `pipeline.whisperx.timeout` | duration | No | `30m` |
| `pipeline.whisperx.retries` | int | No | `3` |
| `pipeline.whisperx.retry_delay` | duration string | No | `2s` |
| `pipeline.whisperx.retry_delay` | duration | No | `2s` |
| `pipeline.whisperx.concurrency` | int | No | `2` |
| `pipeline.seriatim.binary` | string | No | `seriatim` |
| `pipeline.seriatim.timeout` | duration string | No | `10m` |
| `pipeline.seriatim.timeout` | duration | No | `10m` |
| `pipeline.seriatim.output_schema` | string | No | `seriatim-intermediate` |
| `pipeline.seriatim.coalesce_gap` | float | No | `3.0` |
| `pipeline.seriatim.report` | bool | No | `true` |
@@ -177,7 +172,7 @@ Operational notes:
| `pipeline.seriatim.env.backchannel_max_duration` | float | No | unset |
| `pipeline.seriatim.env.filler_max_duration` | float | No | unset |
| `pipeline.audita.binary` | string | No | `audita` |
| `pipeline.audita.timeout` | duration string | No | `3h` |
| `pipeline.audita.timeout` | duration | No | `3h` |
| `pipeline.audita.llm_api_key_env` | string | No | empty |
| `pipeline.audita.modules[]` | list[string] | No | empty |
| `pipeline.audita.base_url` | string | No | empty |
@@ -191,140 +186,124 @@ Operational notes:
| `pipeline.audita.output_schema` | string | No | empty |
| `pipeline.audita.work_dir_retention` | string | No | empty |
| `pipeline.audita.report` | bool | No | `true` |
| `pipeline.normalize.output_path` | string | No | `transcripts/normalized.json` |
| `pipeline.normalize.output_path` | string | No | `transcripts/final.json` |
| `pipeline.normalize.output_schema` | string | No | `seriatim-intermediate` |
| `pipeline.normalize.report` | bool | No | `true` |
| `pipeline.trim.enabled` | bool | No | `false` |
| `pipeline.trim.output_path` | string | Conditional | none |
| `pipeline.trim.bounds.prompt_id` | string | Conditional | none |
| `pipeline.trim.enabled` | bool | No | `true` |
| `pipeline.trim.output_path` | string | No | `transcripts/final.trimmed.json` |
| `pipeline.trim.bounds.prompt_id` | string | No | `dnd.session_bounds` |
| `pipeline.trim.bounds.profile_id` | string | No | empty |
| `pipeline.trim.bounds.transcript_input_name` | string | Conditional | none |
| `pipeline.trim.bounds.output_path` | string | Conditional | none |
| `pipeline.trim.bounds.timeout` | duration string | No | `10m` |
| `pipeline.trim.bounds.transcript_input_name` | string | No | `transcript` |
| `pipeline.trim.bounds.output_path` | string | No | `artifacts/session_bounds.json` |
| `pipeline.trim.bounds.timeout` | duration | No | `10m` |
| `pipeline.trim.bounds.render_debug` | bool | No | `false` |
| `pipeline.trim.bounds.render_output_path` | string | Conditional | none |
| `pipeline.trim.bounds.render_output_path` | string | Conditional | required when `render_debug` is true |
| `pipeline.trim.seriatim.report` | bool | No | `false` |
| `pipeline.notarius.enabled` | bool | No | `false` |
| `pipeline.notarius.binary` | string | No | `notarius` |
| `pipeline.notarius.config_path` | string | Conditional | required when enabled; relative paths resolve from the pipeline file directory |
| `pipeline.notarius.pipeline_id` | string | Conditional | required when enabled |
| `pipeline.notarius.timeout` | duration | No | `3h`; must be positive |
| `pipeline.notarius.working_directory` | string | No | directory containing resolved `config_path`; relative paths resolve from the pipeline file directory |
| `pipeline.notarius.outputs` | map | Conditional | at least one entry when enabled |
| `pipeline.render.enabled` | bool | No | `true` |
| `pipeline.render.format` | string | No | `markdown` (only supported value) |
| `pipeline.render.title` | string | No | empty (falls back to `session.title` when set) |
| `pipeline.render.include_timestamps` | bool | No | `true` |
| `pipeline.render.include_segment_ids` | bool | No | `true` |
| `pipeline.render.include_metadata` | bool | No | `false` |
| `pipeline.scriptorium.binary` | string | No | `scriptorium` |
| `pipeline.scriptorium.config_path` | string | No | empty |
| `pipeline.scriptorium.timeout` | duration string | No | `10m` |
| `pipeline.scriptorium.timeout` | duration | No | `10m` |
| `pipeline.scriptorium.render_debug` | bool | No | `false` |
| `pipeline.scriptorium.artifacts` | map | No | empty |
| `pipeline.scriptorium.artifacts.<name>.enabled` | bool | No | `false` |
| `pipeline.scriptorium.artifacts.<name>.depends_on[]` | list[string] | No | empty |
| `pipeline.scriptorium.artifacts.<name>.render_debug` | bool | No | unset |
| `pipeline.scriptorium.artifacts.<name>.prompt_id` | string | Conditional | none |
| `pipeline.scriptorium.artifacts.<name>.profile_id` | string | No | empty |
| `pipeline.scriptorium.artifacts.<name>.output_path` | string | Conditional | none |
| `pipeline.scriptorium.artifacts.<name>.timeout` | duration string | No | empty |
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source` | string | Conditional | none |
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.artifact` | string | No | empty |
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.path` | string | No | empty |
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.required` | bool | No | `false` |
| `pipeline.scriptorium.artifacts.<name>.vars.<key>` | map value | No | empty |
| `pipeline.analyzer.binary_path` | string | No | empty |
| `pipeline.analyzer.timeout` | duration string | No | empty |
| `pipeline.analyzer.artifacts.output_dir` | string | No | empty |
| `pipeline.analyzer.artifacts.types[]` | list[string] | No | empty |
| `pipeline.notification.backend` | string | No | empty |
| `pipeline.notification.recipient` | string | No | empty |
| `pipeline.notification.timeout` | duration string | No | empty |
| `pipeline.notification.timeout` | duration | No | empty |
Scriptorium artifact-key and dependency rules:
### Notarius Output Entries
- artifact keys must match `^[a-z][a-z0-9_]*$`.
- enabled artifacts require `prompt_id` and `output_path`.
- `output_path` must be relative, traversal-safe, and under `artifacts/`.
- configured artifact input sources use `narratio.artifact.<name>`.
- if input source references `narratio.artifact.<name>`, artifact `<name>` must exist and must be listed in `depends_on`.
- every `depends_on` entry must be a configured artifact key.
- self-dependency is rejected.
- enabled dependency cycles are rejected.
- any artifact referenced by `depends_on` or `narratio.artifact.<name>` source must define `output_path` (even if not enabled).
For each `pipeline.notarius.outputs.<name>`:
Allowed `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source` values:
- `previous_session_artifact`
- `narratio.transcript.merged`
- `narratio.transcript.polished`
- `narratio.transcript.full`
- `narratio.transcript.trimmed`
- `narratio.bounds.session`
- `narratio.artifact.<configured_artifact_key>`
`pipeline.archive.promote_artifacts[].source` values:
- `narratio.transcript.merged`
- `narratio.transcript.polished`
- `narratio.transcript.full`
- `narratio.transcript.trimmed`
- `narratio.bounds.session`
- `narratio.artifact.<configured_artifact_key>`
Archive promotion destination rules:
- `dest` must be a clean relative path (not absolute, no traversal).
- duplicate `dest` values are rejected.
- if `dest` is omitted:
- built-in sources derive their canonical destination path;
- configured sources derive from `pipeline.scriptorium.artifacts.<name>.output_path`;
- derivation failure is a config validation error.
Restore-related implications:
- restore remote identity requires archive S3 identity to resolve (`pipeline.storage.s3.bucket` and session prefix derivation inputs).
- restore scope considers only committed current state and durable paths (`manifest.json`, `transcripts/**`, `artifacts/**`, optional `audio/**`).
## 8. Full session reference
| Path | Type | Required | Default |
| Field | Type | Required | Rule |
| --- | --- | --- | --- |
| `session.session_id` | string | Yes | none |
| `session.campaign` | string | Yes | none |
| `session.date` | string | No | empty |
| `session.title` | string | No | empty |
| `session.inputs.audio_dir` | string | Conditional | empty |
| `session.inputs.audio_files[]` | list[string] | Conditional | empty |
| `session.inputs.audio_s3.prefix` | string | Conditional | none |
| `session.inputs.speakers_file` | string | Yes | none |
| `session.inputs.autocorrect_file` | string | Yes | none |
| `session.inputs.glossary_file` | string | Yes | none |
| `lane_id` | string | Yes | unique Notarius lane ID |
| `media_type` | string | Yes | exact accepted descriptor media type |
| `schema_id` | string | Yes | exact accepted descriptor schema ID |
| `schema_version` | string | Yes | exact accepted descriptor schema version |
| `module_key` | string | No | exact accepted module key when set |
Audio-source rule:
Output names must match `^[a-z][a-z0-9_]*$` and become selectable sources named
`narratio.extraction.<name>`. Lane IDs must be unique. Every declared output is
required from a successful Notarius result; a missing, rejected, duplicate, or
contract-incompatible lane fails extraction. See the
[complete maintained example](../examples/pipeline.full.annotated.yml) for the
current ten-lane D&D mapping and the [Notarius contract](./integrations/notarius.md)
for compatibility ownership.
- configure exactly one mode:
- `audio_dir`, or
- `audio_files` (at least one), or
- `audio_s3.prefix`
- `audio_s3` cannot be combined with local audio fields.
### Scriptorium Artifact Entries
## 9. Secrets
For each `pipeline.scriptorium.artifacts.<name>`:
Narratio supports filesystem-based secret injection via `pipeline.secrets.env_dir`.
| Field | Type | Required | Rule |
| --- | --- | --- | --- |
| `enabled` | bool | No | `false` if omitted |
| `depends_on[]` | list[string] | No | must reference configured artifact keys; no self-reference; enabled graph must be acyclic |
| `render_debug` | bool | No | per-artifact override |
| `prompt_id` | string | Conditional | required when artifact is enabled |
| `profile_id` | string | No | empty |
| `output_path` | string | Conditional | required when enabled; also required when referenced by publish/output/input rules |
| `timeout` | duration | No | artifact override |
| `inputs` | map | No | input key names must be non-empty |
| `vars` | map | No | values must be string or bool; `session_id` is reserved and overwritten by Narratio |
Behavior:
Narratio adds `session_id=narratio-session-<session_id>` to every Scriptorium request for sticky upstream LLM routing. If an artifact config sets `vars.session_id`, Narratio replaces that value before invoking Scriptorium. Use a different variable name if a prompt needs the raw Narratio session ID as content.
- `env_dir` may be absolute or relative.
- relative `env_dir` resolves from current working directory.
- files with valid env-var names (`[A-Za-z_][A-Za-z0-9_]*`) are loaded.
- values are loaded from file contents with trailing newline trimming.
- existing process env vars are preserved.
- invalid names and subdirectories are skipped.
- missing/unreadable `env_dir` fails command execution.
For each artifact input `pipeline.scriptorium.artifacts.<name>.inputs.<input_name>`:
Guidance:
| Field | Type | Required | Rule |
| --- | --- | --- | --- |
| `source` | string | Yes | built-in runtime source, prepared input source, `narratio.extraction.<name>`, `narratio.artifact.<name>`, or `narratio.previous_session.artifact.<name>` |
| `artifact` | string | No | optional passthrough adapter field |
| `path` | string | No | optional passthrough adapter field |
| `required` | bool | No | optional input requirement |
- do not put secret values directly in YAML.
- configure env var names in config and provide values via env/secrets files.
### Campaign
## 10. Examples
| Field | Type | Required | Notes |
| --- | --- | --- | --- |
| `campaign_id` | string | Yes | canonical campaign identity |
| `session_template_file` | string | No | used by `session init` when set |
| `inputs.speakers_file` | string | Yes | stable input default |
| `inputs.autocorrect_file` | string | Yes | stable input default |
| `inputs.glossary_file` | string | Yes | stable input default |
| `inputs.players_file` | string | Yes | stable input default |
| `inputs.party_file` | string | Yes | stable input default |
Maintained examples:
### Session
- `examples/pipeline.minimal.yml`
- `examples/pipeline.production.yml`
- `examples/pipeline.full.annotated.yml`
- `examples/session.template.yml`
- `examples/session.local-audio.yml`
- `examples/session.s3-audio.yml`
| Field | Type | Required in session file | Notes |
| --- | --- | --- | --- |
| `session_id` | string | Yes | must match CLI session target when provided |
| `previous_session_id` | string | No | must not equal `session_id` |
| `campaign` | string | No | filled from `campaign_id` during resolve if omitted |
| `date` | string | No | metadata |
| `title` | string | No | metadata |
| `inputs.speakers_file` | string | No | overrides campaign stable input |
| `inputs.autocorrect_file` | string | No | overrides campaign stable input |
| `inputs.glossary_file` | string | No | overrides campaign stable input |
| `inputs.players_file` | string | No | overrides campaign stable input |
| `inputs.party_file` | string | No | overrides campaign stable input |
| `inputs.audio_dir` | string | Conditional | local audio mode |
| `inputs.audio_files[]` | list[string] | Conditional | local audio mode |
| `inputs.audio_s3.prefix` | string | Conditional | S3 audio mode |
These examples are validated by `internal/config` tests.
Audio rules:
- configure local mode (`audio_dir` or `audio_files`) or S3 mode (`audio_s3.prefix`), not both.
## Maintained Examples
See the [maintained examples index](../examples/README.md) for complete pipeline,
campaign, session, template, and input fixtures. Keep complete copyable files
there rather than duplicating them in this reference.

View File

@@ -1,92 +1,43 @@
# Development Guide
# Development
## Purpose
Canonical contributor workflow and engineering conventions for implemented Narratio behavior.
This is the first-read landing page for people and LLM coding agents working on
Narratio. It provides a concise repository orientation and routes each kind of
change to its canonical documentation.
## Repository layout
Narratio is a stage-driven Go orchestrator for turning D&D session audio into
polished transcripts and generated artifacts. Start with the
[README](../README.md) for product context,
[Architecture](policy/architecture.md) for normative system boundaries, and the
[Internal Overview](internal/overview.md) for implemented component ownership.
- `cmd/narratio/`: CLI entrypoint.
- `internal/app/`: command handlers, plan/run/resume orchestration, cleanup gates, secrets loading.
- `internal/config/`: strict YAML loading, defaults, and validation.
- `internal/stage/`: stage implementations and stage registry/order.
- `internal/adapters/`: external boundary adapters (WhisperX, Seriatim, Audita, Scriptorium, storage, notify).
- `internal/manifest/`: session/run manifest types and persistence.
- `internal/artifacts/`: canonical local/remote path helpers and local artifact store.
- `docs/`: canonical documentation set.
- `examples/`: maintained config examples used by tests.
## What To Read
## Build and test commands
| When working on | Read | Why |
| --- | --- | --- |
| Finding the package or component that owns current behavior | [Internal Overview](internal/overview.md) | It is the implemented component inventory and routes to focused internal documents. |
| Application shape, boundaries, dependency direction, runtime invariants, safety properties, or dependencies | [Architecture](policy/architecture.md) | It defines the intended system shape, ownership, and non-goals. |
| Any documentation addition or revision | [Documentation Policy](policy/documentation.md) | It defines canonical owners, audiences, current-behavior rules, and maintenance requirements. |
| Adding, changing, reviewing, rewriting, or deleting tests | [Testing Policy](policy/testing.md) | It defines risk-based sufficiency, durable test boundaries, test-double guidance, and test lifecycle decisions. |
| CLI composition or command behavior | [Internal Overview](internal/overview.md) and [CLI Reference](cli.md) | The overview routes to command ownership; the reference owns public syntax and invocation behavior. |
| Configuration loading, resolution, or user-visible configuration | [Internal Overview](internal/overview.md) and [Configuration](config.md) | The overview routes to implementation ownership; the reference owns fields, defaults, discovery, and validation. |
| Session workflow, status, restore, cleanup, or object storage | [Restore Internals](internal/command-restore.md), [Workspace Internals](internal/workspace.md), [Storage Internals](internal/storage.md), [Operations](operations.md), and [Troubleshooting](troubleshooting.md) | These separate implementation mechanics, operator procedures, and symptom-driven recovery. |
| Pipeline sequencing or the behavior of a stage | [Internal Overview](internal/overview.md) and its focused stage documents | The overview owns the implemented stage inventory and routes to each stage contract. |
| Adapters or external tool contracts | [Adapter Internals](internal/adapters.md) and [Integration Contracts](integrations/README.md) | The internal guide owns adapter composition and mechanics; integration documents own external formats and protocols. |
| Manifests, artifacts, workspace paths, or publish behavior | [Manifest Internals](internal/manifest.md), [Artifact Internals](internal/artifacts.md), [Workspace Internals](internal/workspace.md), [Publish Internals](internal/stage-publish.md), and [Operations](operations.md) | These separate implementation state and resolution from operator-visible layout and lifecycle. |
| Maintained configuration or input examples | [Configuration](config.md) and [Examples](../examples/README.md) | The reference owns field meanings; the examples directory owns complete copyable files. |
| Proposed or unimplemented behavior | [Roadmap](roadmap/) | Future work belongs only in roadmap documentation until implemented. |
- Run focused CLI behavior checks:
For an existing subsystem, also inspect its focused tests and package-level
contracts before changing behavior.
```bash
go test ./internal/app -run TestExecute -v
```
## Validation
- Run config example load/validate checks:
Use focused package tests while iterating. Run the repository-wide checks when a
change affects shared contracts, application behavior, or maintained
documentation examples:
```bash
go test ./internal/config -run TestExamplesLoadAndValidate -v
```
- Run full test suite:
```bash
```sh
go test ./...
go vet ./...
go build ./cmd/narratio
```
## Coding conventions
- Keep orchestration explicit and stage-driven; do not introduce generic workflow/DAG abstractions.
- Keep external-system details inside adapter packages; stages should consume Narratio-level contracts only.
- Use centralized path helpers from `internal/artifacts` rather than ad hoc path concatenation.
- Preserve manifest-driven state transitions (`running`, `succeeded`, `failed`, `skipped`, `stale`) as the source of run progress.
- Keep user/operator docs implementation-accurate; planned work belongs only under `docs/roadmap/`.
For design principles and invariants, see [docs/architecture.md](./architecture.md). For stage/adapter contracts, see [docs/internal/README.md](./internal/README.md).
## Dependency policy
- Prefer Go standard library where practical.
- Add third-party dependencies only when they provide clear value for required behavior.
- Keep dependency additions narrow to the boundary package that needs them.
## Change playbooks
### Add config fields
1. Add fields to config structs in `internal/config`.
2. Set defaults in `internal/config/defaults.go` when appropriate.
3. Add validation rules in `internal/config/validate.go`.
4. Add or update load/validate tests in `internal/config/*_test.go`.
5. Update canonical config docs and examples:
- [docs/config.md](./config.md)
- relevant files under `examples/`
### Add CLI flags or commands
1. Update command parsing and behavior in `internal/app`.
2. Add or update command tests (`TestExecute` and command-specific tests).
3. Update [docs/cli.md](./cli.md) and, if operator workflow changes, [docs/operations.md](./operations.md).
### Add or modify stages/adapters
1. Implement stage behavior in `internal/stage` with clear input/output boundaries.
2. Keep external transport/subprocess details in `internal/adapters`.
3. Preserve manifest and promotion semantics expected by runner and archive logic.
4. Add/update stage and adapter tests.
5. Update internal component contracts in `docs/internal/`.
### Update examples
1. Keep canonical examples only in `examples/`.
2. Ensure examples load and validate through runtime config paths.
3. Update `internal/config/load_validate_test.go` as needed.
4. Update links in `docs/config.md` if example filenames change.
### Update docs and roadmap
1. Keep implemented behavior in canonical docs (`README`, `docs/*.md`, `docs/internal/`).
2. Keep planned/unimplemented behavior only in `docs/roadmap/`.
3. After completing roadmap items, remove or mark them complete in `docs/roadmap/documentation.md`.
4. Run a link/path sweep before finalizing changes.

View File

@@ -1,356 +0,0 @@
# Go Project Documentation Policy
## Purpose
Project documentation must help four audiences:
1. users who need to run the application;
2. administrators/operators who need to configure and operate it;
3. developers who need to understand and change it safely;
4. LLM coding agents that need clear scope, boundaries, and invariants.
Docs should be accurate, concise, task-oriented, and organized by audience. Prefer links to canonical docs over repetition.
## Core Rules
### 1. Keep docs concise
Each document should cover a defined scope and only the essentials for that scope.
Avoid:
- long background explanations;
- repeated reference material;
- implementation detail in user-facing docs;
- aspirational language outside roadmap docs;
- verbose examples where one minimal example is clearer.
### 2. Document only implemented behavior outside roadmap files
Unimplemented, planned, aspirational, experimental, or future work may be described only under:
- `docs/roadmap/`
No other documentation file, including `README.md`, should describe code, features, modules, stages, commands, config fields, or behaviors that do not currently exist.
If a feature is partial, non-roadmap docs may describe only the implemented portion and its current boundary.
### 3. Use canonical homes
Each type of information should have one canonical location.
Canonical homes:
- project purpose and quickstart: `README.md`
- development principles: `docs/architecture.md`
- configuration reference: `docs/config.md`
- CLI reference: `docs/cli.md`
- operations and recovery: `docs/operations.md`
- troubleshooting: `docs/troubleshooting.md`
- implemented internals: `docs/internal/`
- future work: `docs/roadmap/`
- contributor workflow: `docs/development.md`
- copyable examples: `examples/`
Other files should summarize briefly and link to the canonical source.
### 4. Keep examples real
Examples should be valid, maintained, and free of secrets.
Where practical:
- example configs should load successfully;
- example commands should match real CLI syntax;
- important examples should be covered by tests.
## Documentation Profiles
All projects require:
- `README.md`
- `docs/architecture.md`
Additional docs depend on the project.
### Small library
Recommended:
- `docs/development.md`, if contributor conventions are non-obvious
### Simple CLI
Required:
- `docs/cli.md`
Recommended:
- `docs/development.md`
### Config-driven CLI
Required:
- `docs/cli.md`
- `docs/config.md`
Recommended:
- `examples/`
- `docs/development.md`
### Stateful or operator-facing application
Required:
- `docs/cli.md`, if CLI-based
- `docs/config.md`, if config-driven
- `docs/operations.md`
Recommended:
- `docs/troubleshooting.md`
- `examples/`
- `docs/development.md`
### Modular, staged, service-oriented, or orchestration application
Required:
- `docs/cli.md`, if CLI-based
- `docs/config.md`, if config-driven
- `docs/operations.md`
- `docs/internal/`
- `docs/development.md`
Recommended:
- `docs/troubleshooting.md`
- validated examples under `examples/`
## Required Documents
### README.md
**Audience:** users, administrators, operators
The README is the outward-facing project orientation page.
It should include, in order:
1. concise description;
2. elevator pitch;
3. shortest useful command or usage example;
4. links to targeted docs.
The README should be short. It is not a manual.
The “shortest useful command” means the simplest command that performs the projects core use case. (It does not mean `app --help`.)
### docs/architecture.md
**Audience:** developers, LLM coding agents
`docs/architecture.md` is required for every project.
It is an inward-facing development policy document. It should describe how the project is intended to be built and changed.
It should include:
- project shape;
- core design principles;
- package and boundary philosophy;
- state/persistence philosophy, if applicable;
- external integration philosophy, if applicable;
- error-handling and logging principles;
- testing expectations;
- documentation expectations;
- architectural invariants;
- explicit non-goals, if useful.
For small projects, this file may be brief. It may simply state that the project is intentionally narrow, monolithic, and dependency-light.
### docs/config.md
**Audience:** administrators, operators, advanced users
Required for applications with configuration files.
It should include, in order:
1. config file locations and discovery precedence;
2. minimal working config;
3. production-oriented config;
4. full configuration reference;
5. secrets handling, if applicable;
6. links to maintained examples.
The full configuration reference should be canonical.
### docs/cli.md
**Audience:** users, administrators, operators
Required for CLI applications.
It should include, in order:
1. shortest useful command;
2. command overview;
3. complete flag reference;
4. common workflows;
5. diagnostic or recovery commands, if applicable.
Explain when commands are useful, not just their syntax.
### docs/operations.md
**Audience:** administrators, operators
Required for applications that maintain state, support resume behavior, run multiple stages, write durable artifacts, use remote storage, or require recovery procedures.
It should cover:
- normal workflow;
- filesystem layout;
- remote storage layout, if applicable;
- logs and manifests;
- resume/retry behavior;
- cleanup behavior;
- archive/backup behavior;
- safe recovery procedures;
- operational caveats.
### docs/troubleshooting.md
**Audience:** administrators, operators
Recommended once recurring failure modes exist.
Each entry should include:
- symptom;
- likely cause;
- diagnostic command or inspection step;
- safe fix;
- relevant links.
### docs/development.md
**Audience:** developers, LLM coding agents
Required for projects maintained by humans and LLM coding agents.
It should include:
- repository layout;
- build/test commands;
- coding conventions;
- dependency policy;
- how to add config fields;
- how to add CLI flags;
- how to add stages/modules/adapters, if applicable;
- how to update examples;
- documentation update expectations.
### docs/internal/
**Audience:** developers, LLM coding agents
Required for modular, staged, service-oriented, or orchestration projects.
This directory describes implemented internal components. It is not the roadmap.
Use one file per major component where useful.
Each component doc should include:
1. purpose;
2. inputs and outputs;
3. boundaries;
4. config fields used;
5. external adapters used;
6. state or manifest behavior, if applicable;
7. skip/resume behavior, if applicable;
8. failure behavior;
9. tests to inspect before changing;
10. architectural invariants.
### docs/roadmap/
**Audience:** maintainers, developers, LLM coding agents
This is the only place for planned, future, aspirational, experimental, or unimplemented work.
Roadmap docs should clearly distinguish:
- proposed work;
- accepted plans;
- deferred ideas;
- rejected ideas;
- implementation prompts or task breakdowns, if useful.
Roadmap docs should not be confused with current behavior.
### docs/integrations/
**Audience:** developers, LLM coding agents
Required for projects that depend on external CLIs, APIs, services, protocols, or file formats where the integration contract is important to maintain.
This directory contains concise, versioned reference notes for external integration contracts. It should document only the parts of the external system that this project actually uses.
Use one file per integration where useful.
## Examples Directory
Projects with non-trivial configuration or workflows should include `examples/`.
Useful examples include:
- minimal working config;
- production-oriented config;
- full annotated config;
- local development config;
- remote/object-storage config;
- minimal session/input file.
Examples should be valid, maintained, tested when practical, and linked from relevant docs.
## Security and Privacy
Docs and examples must not include:
- real API keys;
- tokens;
- passwords;
- private keys;
- private environment dumps;
- sensitive user data;
- raw private transcripts;
- private infrastructure details unless intentionally public.
Document secret-handling mechanisms, not actual secret values.
## Maintenance Rules
When docs change, verify the affected behavior.
Where practical:
- load example config files in tests;
- test CLI examples or command parser behavior;
- validate documented flags against real flags;
- remove stale references;
- update links after renames;
- keep roadmap content out of non-roadmap docs.
If documentation and code disagree, fix the documentation and/or open a roadmap item; do not leave aspirational behavior in current-behavior docs.
Documentation is complete only when it matches the current code.
## Documentation Change Checklist
Before merging documentation changes, verify:
- README is concise and orientation-focused.
- `docs/architecture.md` describes development principles.
- Future work appears only under `docs/roadmap/`.
- User-facing docs avoid unnecessary internals.
- Developer-facing docs preserve boundaries and invariants.
- Config examples match the schema.
- CLI examples match real commands and flags.
- Defaults appear in the canonical config reference.
- No secrets or private data are included.
- Links are accurate.

View File

@@ -1,15 +1,35 @@
# Integration Documentation Index
# Integrations Index
## Audience
Developers and LLM coding agents changing Narratio's external integration contracts.
Operators, developers, and coding agents who need to understand Narratio's
externally observable integration boundaries.
## Scope
Implemented-only reference notes for the external systems Narratio currently integrates with.
## Integration Docs
- `audita.md`: Audita adapter invocation and validation contract.
- `seriatim.md`: Seriatim normalize/merge/trim adapter contract.
- `scriptorium.md`: Scriptorium run/render adapter contract.
`docs/integrations/` is the canonical reference for protocols, invocation and
data contracts, logical outputs, and compatibility behavior at external tool
boundaries.
## Canonical Owner
`docs/integrations/` is the canonical home for external integration reference notes per `docs/documentation/policy.md`.
These documents describe what Narratio sends or invokes, what it accepts in
return, and how failures are surfaced. Internal composition and stage mechanics
belong in [the adapter implementation guide](../internal/adapters.md) and the
focused stage documents.
## Integration Contracts
- [Audita](./audita.md): transcript polishing (`audita process`).
- [Notarius](./notarius.md): complete pipeline execution and safe JSON bundle
discovery (`notarius run`).
- [Seriatim](./seriatim.md): merge, normalize, trim, and render operations.
- [Scriptorium](./scriptorium.md): artifact generation and debug rendering
(`scriptorium run|render`).
- [WhisperX](./whisperx.md): speaker-audio transcription over HTTP.
## Related Canonical Docs
- [Configuration](../config.md): operator-facing configuration reference.
- [Adapter implementation](../internal/adapters.md): shared adapter boundary and
runner wiring.
- [Internal documentation](../internal/overview.md): stage-specific integration
usage and component ownership.

View File

@@ -1,66 +1,59 @@
# Integration: audita
# Integration: Audita
## Purpose
Define Narratio's adapter contract for transcript polishing via Audita CLI subprocess execution.
Define the Audita adapter contract used by the `polish` stage.
## Inputs and Outputs
Inputs (`audita.PolishRequest`):
- merged transcript path
- glossary path
- output processed transcript path
- optional report path (required when report enabled)
- work dir
- generated config path
- stdout/stderr log paths
- optional module/model/base URL and concurrency knobs
## External Boundary
Outputs (`audita.PolishResult`):
- processed transcript path
- optional report path
- generated config path
- stdout/stderr log paths
- exit code, duration, invoked binary
- adapter metadata
Narratio invokes `audita process` as a subprocess for each polish operation.
The configured timeout and parent cancellation bound the invocation. Internal
runner composition is documented in
[the adapter implementation guide](../internal/adapters.md).
## Boundaries
Owns:
- Deterministic CLI argument construction for `audita process`
- Environment bridging for API credentials
- Invocation config emission
- Output validation for processed transcript and report
## Request Contract
`PolishRequest` carries:
- required transcript/glossary/output/work-dir paths;
- optional report path (required when report mode is enabled);
- generated config and stdout/stderr log paths;
- optional module/model/base-url/config/output-schema/concurrency settings.
Does not own:
- Upstream/downstream stage orchestration
- Credential sourcing policy beyond required env-var presence check
## Result Contract
`PolishResult` returns:
- processed transcript path;
- optional report path;
- work dir and generated-config/log paths;
- exit code, duration, binary provenance;
- adapter metadata map.
## Config Fields Used
Via `pipeline.audita.*` mapped in app/stage wiring:
- `binary`, `timeout`, `llm_api_key_env`, `modules`, `base_url`, `model`
- `transcript_description`, `config_path`, `output_schema`, `work_dir_retention`
- `total_llm_concurrency`, `proposal_llm_concurrency`, `validation_model`, `validation_llm_concurrency`, `report`
## Validation and Failure Semantics
Construction fails for invalid static config values, including:
- empty binary;
- non-positive timeout;
- invalid base URL;
- invalid output schema;
- invalid work-dir retention value;
- invalid concurrency values.
## External Adapters Used
- Shared subprocess helper (`internal/adapters/subprocess`) to run CLI and capture logs.
Run fails for:
- missing required request paths;
- missing required credential env var when configured (`llm_api_key_env`);
- subprocess execution failure;
- invalid processed transcript JSON (`segments` array required);
- invalid report JSON when reporting is enabled.
## State and Manifest Behavior
- No direct manifest writes.
- Stage-level metadata records adapter provenance and credential-present signal.
- Generated invocation YAML is written when `GeneratedConfigPath` is provided.
Failure results still include output/log/config/exit metadata for diagnostics.
## Skip and Resume Behavior
- Adapter has no skip/resume logic. Stage/runner controls this.
## Deterministic Behavior
- CLI args are built from runner config + request in a fixed order.
- Generated invocation YAML (`audita.generated.v1`) is emitted when requested.
- Manifest writes are stage-owned; adapter itself is stateless.
## Failure Behavior
- Constructor validation fails on invalid binary/timeout/schema/concurrency/URL values.
- Run fails on missing required paths, missing required credential env var, subprocess errors, invalid processed JSON shape, or invalid report JSON.
- Failures preserve stdout/stderr paths in returned result metadata.
## Configuration
## Tests to Inspect Before Changing
- `internal/adapters/audita/subprocess_test.go`
- `internal/adapters/audita/fake_test.go`
- `internal/stage/polish_test.go`
Operator-selected values are defined under `pipeline.audita.*` in the
[configuration reference](../config.md#pipeline).
## Architectural Invariants
- Processed output must be valid JSON with top-level `segments` array.
- When report is enabled, report output must be valid JSON.
- If `llm_api_key_env` is configured, credential must be present in environment.
Maintained example with Audita config:
- [Full annotated pipeline](../../examples/pipeline.full.annotated.yml)
- [Production-shaped pipeline](../../examples/pipeline.production.yml)

View File

@@ -0,0 +1,83 @@
# Notarius Integration Contract
## Boundary
Narratio uses Notarius as a subprocess to extract configured structured JSON
lanes from the final trimmed Seriatim transcript. Narratio owns invocation,
safe bundle discovery, lane selection, and its own artifact metadata. Notarius
owns pipeline definitions, lane schemas, the receipt, and bundle formats.
Canonical Notarius references:
- [Subprocess consumer contract](https://gitea.maximumdirect.net/eric/notarius/src/branch/main/docs/consumers/subprocess.md)
- [D&D pipeline and lane contracts](https://gitea.maximumdirect.net/eric/notarius/src/branch/main/docs/consumers/dnd-pipeline.md)
- [Run-result receipt](https://gitea.maximumdirect.net/eric/notarius/src/branch/main/docs/integrations/run-result.md)
- [JSON output bundle](https://gitea.maximumdirect.net/eric/notarius/src/branch/main/docs/integrations/json-output.md)
The [complete Narratio example](../../examples/pipeline.full.annotated.yml)
records the exact current constraints for all ten D&D lanes. Treat the linked
Notarius documents as canonical when changing those values; Narratio does not
duplicate the complete schemas.
## Invocation
When `pipeline.notarius.enabled` is true, Narratio resolves the executable,
configuration path, input path, output directory, and working directory to
absolute paths and invokes:
```text
notarius run <pipeline_id> --config <config_path> --input <trimmed_json> --output-dir <staging_dir> --json
```
Standard output is reserved for the JSON receipt. Standard error is captured
separately as diagnostic output. Narratio applies the configured timeout and
does not interpret stdout as a receipt unless the subprocess exits successfully.
It does not pass a Narratio session ID or run `notarius config validate`
automatically; the configured working directory and inherited environment
apply to the subprocess.
## Accepted Result
Narratio currently accepts receipt schema `notarius.run-result.v1`. The receipt
must identify the configured pipeline, and its `index_file` must be exactly
`index.json` beneath the reported bundle root. The production index must name
the management files exactly as `manifest.json`, `rejected.json`, and
`warnings.json`. All receipt, index, and lane paths must stay inside that
bundle; symlinks and non-regular lane payloads are rejected.
Supported receipt and index shapes tolerate unknown fields for forward
compatibility, while required identity, validation, count, manifest,
rejection, warning, and lane-list fields remain mandatory. Narratio applies
bounded reads to the receipt, index, rejection, and warning documents. Optional
chunk-map and evidence-context descriptors must carry their complete generic
contract metadata when present.
For every entry in `pipeline.notarius.outputs`, Narratio requires exactly one
index descriptor with the configured lane ID, media type, schema ID, schema
version, and, when configured, module key. Missing, duplicate, rejected, or
incompatible required lanes fail extraction even if Notarius exited zero.
Unconfigured lanes may remain in the preserved bundle but do not become
selectable Narratio sources.
Each accepted configured lane is registered as
`narratio.extraction.<output_key>`. The bundle index is retained for audit and
resume validation but is not selectable. Scriptorium and publish rules consume
only explicitly named lane sources; `--artifacts` never selects Notarius lanes.
## Failure And Compatibility Behavior
- Startup and nonzero-exit errors fail extraction and retain captured diagnostics.
- Invalid receipt JSON or an unsupported receipt schema fails before bundle use.
- Unsafe or incompatible index data and required-lane rejection fail before the
staged bundle is promoted to durable storage.
- Contract and external provenance metadata are preserved on lane artifact
records and through explicit publication.
Rejection and warning summaries retain structured stage, scope, lane, and
reason-code fields for diagnostics without exposing free-form external messages
or reading lane payload bodies.
Configuration fields and defaults are in [Configuration](../config.md).
Operator paths, rerun procedures, and bundle retention are in
[Operations](../operations.md). See [Troubleshooting](../troubleshooting.md)
for failure recovery.

View File

@@ -1,64 +1,68 @@
# Integration: scriptorium
# Integration: Scriptorium
## Purpose
Define Narratio's adapter contract for Scriptorium artifact generation and render-debug subprocess invocations.
Define the Scriptorium adapter contract used by `analyze` and trim-bounds generation in `trim`.
## Inputs and Outputs
Inputs:
- `RunArtifactRequest`: binary, config path, prompt/profile IDs, input map, vars map, timeout, output path, logs/config paths, optional API env and working dir
- `RenderArtifactRequest`: same core fields for render mode
## External Boundary
Outputs (`ArtifactResult`):
- output path
- stdout/stderr log paths
- generated config path
- exit code and duration
- command mode (`run` or `render`)
- prompt/profile provenance
- validation failure signal
- adapter metadata
Narratio invokes Scriptorium as a subprocess in these modes:
## Boundaries
Owns:
- Deterministic CLI arg construction for `scriptorium run` and `scriptorium render`
- Common request validation
- Invocation config emission
- Output existence/non-empty checks
- Validation-failure mapping for run exit code 2
- `scriptorium run`
- `scriptorium render`
Does not own:
- Artifact selection policy (`analyze` stage)
- Bounds semantic validation (`trim` stage)
The request timeout and parent cancellation bound each invocation. Internal
runner composition is documented in
[the adapter implementation guide](../internal/adapters.md).
## Config Fields Used
Via `pipeline.scriptorium.*` and stage-level artifact config:
- `binary`, `config_path`, `timeout`, `render_debug`
- artifact-level `prompt_id`, `profile_id`, `timeout`, `inputs`, `vars`, `output_path`
## Request Contract
Both request types carry:
- binary/config/prompt/profile IDs;
- input map and vars map;
- output path;
- timeout;
- generated config + stdout/stderr log paths;
- optional API-key env var name;
- optional working directory.
## External Adapters Used
- Shared subprocess helper (`internal/adapters/subprocess`).
## Result Contract
`ArtifactResult` returns:
- output/log/generated-config paths;
- exit code and duration;
- command mode (`run` or `render`);
- prompt/profile provenance;
- `ValidationFailed` marker;
- metadata map.
## State and Manifest Behavior
- No direct manifest writes.
- Stage metadata records adapter outputs and command mode.
- Generated invocation YAML is written when requested.
## Validation and Failure Semantics
Request validation fails for:
- missing binary, prompt id, or output path;
- non-positive timeout;
- empty input/var names;
- empty input path values;
- missing required credential env var when `APIKeyEnv` is set.
## Skip and Resume Behavior
- Adapter has no skip/resume logic. Stage/runner controls execution.
Run behavior:
- subprocess errors propagate with context;
- `run` exit code `2` is mapped to `ValidationFailed=true`;
- successful subprocess still fails if output file is missing or empty.
## Failure Behavior
- Request validation fails for missing binary/prompt/output, invalid timeout, invalid input/var names, or missing required API env var.
- Subprocess errors bubble with command context.
- `run` exit code 2 is treated as `ValidationFailed=true` and surfaced as error by calling stage.
- Successful subprocess still fails if output file is missing/empty.
Render behavior:
- subprocess errors propagate;
- output file must exist and be non-empty.
## Tests to Inspect Before Changing
- `internal/adapters/scriptorium/subprocess_test.go`
- `internal/adapters/scriptorium/fake_test.go`
- `internal/stage/analyze_test.go`
- `internal/stage/trim_test.go`
## Deterministic Behavior
- input and var maps are sorted into deterministic `--input` and `--var` CLI args.
- stage wiring adds `session_id=narratio-session-<session_id>` to every Scriptorium request for sticky upstream routing, overriding any configured `vars.session_id`.
- generated invocation YAML (`scriptorium.generated.v1`) is emitted when requested.
- adapter is stateless and does not own artifact-selection policy.
## Architectural Invariants
- Both modes require explicit timeout > 0.
- Input/var maps are sorted into deterministic CLI argument order.
- Run-mode validation failures are represented explicitly, not silently skipped.
## Configuration
Operator-selected values are defined under `pipeline.scriptorium.*`, including
per-artifact settings under `pipeline.scriptorium.artifacts.*`, in the
[configuration reference](../config.md#pipeline).
Maintained examples with Scriptorium config:
- [Full annotated pipeline](../../examples/pipeline.full.annotated.yml)
- [Production-shaped pipeline](../../examples/pipeline.production.yml)

View File

@@ -1,60 +1,60 @@
# Integration: seriatim
# Integration: Seriatim
## Purpose
Define Narratio's adapter contract for merge, normalize, and trim subprocess invocations of Seriatim.
Define the Seriatim adapter contract used by `merge`, `normalize`, `trim`, and `render`.
## Inputs and Outputs
Inputs:
- `MergeRequest`: raw/normalized transcript inputs, output path, optional report, speaker/autocorrect paths, logs/config
- `NormalizeRequest`: input transcript, output path, schema, optional report, timeout/log/config
- `TrimRequest`: input transcript, output path, keep selector, timeout/log/config
## External Boundary
Outputs:
- `MergeResult`, `NormalizeResult`, `TrimResult` with output paths, logs/config paths, exit code, duration, binary provenance, and metadata.
Narratio invokes Seriatim as a subprocess in these modes:
## Boundaries
Owns:
- Validated deterministic CLI invocation construction
- Optional env tuning propagation for merge
- Invocation config file emission
- JSON output validation
- `seriatim merge`
- `seriatim normalize`
- `seriatim trim`
- `seriatim render`
Does not own:
- Transcript input selection/promotion logic (stage-owned)
- Bounds computation (scriptorium/trim-stage-owned)
The configured timeout and parent cancellation bound each invocation. Internal
runner composition is documented in
[the adapter implementation guide](../internal/adapters.md).
## Config Fields Used
Via `pipeline.seriatim.*` mapped in app/stage wiring:
- `binary`, `timeout`, `output_schema`, `coalesce_gap`, `report`
- `env.overlap_word_run_gap`
- `env.overlap_word_run_reorder_window`
- `env.backchannel_max_duration`
- `env.filler_max_duration`
## Request/Result Contracts
- `MergeRequest`/`MergeResult`: multi-input merge to base transcript, optional report.
- `NormalizeRequest`/`NormalizeResult`: transcript normalization with explicit schema.
- `TrimRequest`/`TrimResult`: transcript trimming with required keep selector.
- `RenderRequest`/`RenderResult`: transcript-to-markdown rendering with explicit format and render booleans.
## External Adapters Used
- Shared subprocess helper (`internal/adapters/subprocess`).
Results include output/log/config paths, timing, exit code, and metadata.
## State and Manifest Behavior
- No direct manifest writes.
- Stage metadata consumes adapter result fields and preserves generated config/log references.
## Validation and Failure Semantics
Runner construction validates:
- binary presence;
- timeout > 0;
- supported output schema (`seriatim-minimal|seriatim-intermediate|seriatim-full`);
- non-negative coalesce gap.
## Skip and Resume Behavior
- Adapter has no skip/resume logic. Runner controls stage execution.
Invocation fails on:
- missing required request paths/inputs;
- invalid normalize schema override;
- unsupported render format;
- subprocess failure;
- invalid JSON outputs for merge/normalize/trim;
- missing `segments` array for normalize/trim transcript outputs;
- empty render output files.
## Failure Behavior
- Constructor fails for invalid binary/timeout/output-schema/coalesce-gap.
- Merge fails on missing output path/inputs/report path (if enabled), subprocess errors, invalid merged output JSON, invalid report JSON.
- Normalize fails on missing input/output, invalid schema, subprocess errors, invalid normalized output JSON shape, invalid report JSON.
- Trim fails on missing input/output/keep selector, subprocess errors, invalid trimmed output JSON shape.
When report paths are provided/enabled, report files must parse as JSON.
## Tests to Inspect Before Changing
- `internal/adapters/seriatim/subprocess_test.go`
- `internal/adapters/seriatim/fake_test.go`
- `internal/stage/merge_test.go`
- `internal/stage/normalize_test.go`
- `internal/stage/trim_test.go`
## Deterministic Behavior
- argument ordering is deterministic per command construction.
- merge env overrides are explicit (`SERIATIM_*`) and only emitted when configured.
- generated invocation YAML (`seriatim.generated.v1`) is emitted when requested.
- adapter does not write manifests or choose stage inputs.
## Architectural Invariants
- Supported output schemas are limited to `seriatim-minimal`, `seriatim-intermediate`, `seriatim-full`.
- Normalize/trim outputs must include `segments` arrays.
- Merge/normalize/trim all route through deterministic subprocess invocation.
## Configuration
Operator-selected values are defined under `pipeline.seriatim.*` and
`pipeline.render.*` in the
[configuration reference](../config.md#pipeline).
Maintained examples with Seriatim config:
- [Full annotated pipeline](../../examples/pipeline.full.annotated.yml)
- [Production-shaped pipeline](../../examples/pipeline.production.yml)

View File

@@ -0,0 +1,66 @@
# Integration: WhisperX
## Purpose
WhisperX transcribes each prepared speaker audio file for Narratio's
`transcribe` stage. Narratio uses an HTTP boundary and installs each successful
response as that speaker's raw transcript JSON.
## HTTP Boundary
Narratio sends an HTTP `POST` to the configured transcription URL using
`multipart/form-data` with:
- `file`: the audio file, retaining its base filename; and
- `language`: the configured language string.
The server must return a `2xx` response whose body is valid JSON. Narratio does
not currently require a more specific response schema at this boundary.
## Request And Result Contract
Each adapter request identifies a speaker, a readable audio file, and the
destination for the raw transcript. The HTTP request carries the audio and
language; the speaker identifier remains Narratio orchestration metadata.
On success, Narratio atomically writes the response body to the requested
destination. The adapter result reports that logical output together with the
attempt count, final HTTP status when available, elapsed duration, and adapter
identity metadata. A failed or invalid response is not installed as the
transcript output.
## Retry, Timeout, And Cancellation
- The configured timeout applies independently to each HTTP attempt.
- `retries` means additional attempts after the first.
- HTTP `429`, HTTP `5xx`, attempt timeouts, and network errors are retryable.
- Other HTTP `4xx` responses and explicit cancellation are not retryable.
- Narratio waits the configured retry delay between attempts and aborts that
wait when the parent context is canceled.
## Validation And Failure Semantics
Client construction rejects a missing or invalid absolute transcription URL,
a missing language, a non-positive timeout, negative retries, or a negative
retry delay. A request fails before transmission when its audio or output path
is missing.
Non-`2xx` status, transport failure, response-size overflow, invalid JSON, or
failure to install the output causes the transcription to fail. Errors include
attempt context, and the result retains attempts, final status when available,
and elapsed duration for diagnostics.
## Determinism And Concurrency
Each audio request has stable multipart field names, and successful bytes are
installed atomically. The transcribe stage may process speaker files in
parallel, bounded by the configured concurrency. It records results in stable
speaker order after all work completes; any speaker failure fails the stage.
## Related Canonical Docs
- [Configuration](../config.md#pipeline) defines the operator-selected
WhisperX URL, language, timeouts, retry policy, and concurrency.
- [Adapter implementation](../internal/adapters.md) describes internal wiring.
- [Transcribe stage](../internal/stage-transcribe.md) describes stage mechanics,
durable artifacts, and manifests.

View File

@@ -1,29 +0,0 @@
# Internal Documentation Index
## Audience
Developers and LLM coding agents changing Narratio internals.
## Scope
Implementation-accurate contracts for workspace/state, manifests, stages, artifact resolution, adapter boundaries, and restore command behavior.
## Component Docs
- `adapters.md`: external adapter map, runtime wiring, and boundary ownership.
- `storage.md`: remote storage backend contracts and object-store invariants.
- `manifest.md`: session/run manifest schemas, lifecycle transitions, and persistence semantics.
- `artifacts.md`: built-in artifact registry, runtime artifact catalog, and source-resolution behavior.
- `workspace.md`: local state model, manifests, run-local layout, promotion, and cleanup invariants.
- `command-restore.md`: restore command discovery/planning/execution/reporting contract.
- `stage-prepare.md`: input materialization and provenance capture.
- `stage-transcribe.md`: WhisperX transcript generation.
- `stage-merge.md`: Seriatim normalization + merge.
- `stage-polish.md`: Audita transcript polishing.
- `stage-normalize.md`: post-polish normalization.
- `stage-trim.md`: bounds-driven transcript trimming.
- `stage-analyze.md`: dependency-ordered Scriptorium artifact generation for selected configured artifacts.
- `stage-archive.md`: archive upload and current-pointer publish contract.
## External Integration Notes
- `../integrations/README.md`: canonical location for external integration contracts (`audita.md`, `seriatim.md`, `scriptorium.md`).
## Canonical Owner
`docs/internal/` is the canonical home for implemented internals per `docs/documentation/policy.md`.

View File

@@ -1,79 +1,78 @@
# Internal: Adapters
## Purpose
Describe the external adapter boundaries used by Narratio stages and app orchestration, including default runtime wiring.
## Inputs and outputs
Inputs:
- Stage requests passed through adapter interfaces (for example transcription, merge/normalize/trim, polish, artifact generation, object-store operations, notifications).
- Resolved config values used to construct default adapters.
Explain the adapter interfaces and production composition used by application
and stage orchestration. Externally observable protocols and formats belong in
the [integration contracts](../integrations/).
Outputs:
- Adapter-specific result structs (paths, metadata, status/attempt info, duration/exit details).
- Adapter errors returned to stage/app orchestration.
## Adapter Boundaries
## Boundaries
Owns:
- Transport/process/SDK details at system boundaries (`HTTP`, subprocess CLI invocation, AWS SDK calls).
- Request/response contracts in `internal/adapters/*` packages.
Narratio stage logic depends on adapter interfaces, not transport-specific details.
Does not own:
- Stage sequencing, skip/force/resume decisions.
- Manifest transition logic.
- Canonical workspace path policy.
Primary adapters:
## Config fields used
Default wiring and adapter calls consume:
- `pipeline.whisperx.*`
- `pipeline.seriatim.*`
- `pipeline.audita.*`
- `pipeline.scriptorium.*`
- `pipeline.storage.*` and `pipeline.archive.*` (object-store construction/gating)
- `pipeline.notification.*` (sender boundary exists; placeholder behavior today)
## External adapters used
Runtime env boundary fields (`internal/stage.Env`):
- `whisperx.Client`
- `seriatim.Runner`
- `audita.Runner`
- `scriptorium.Runner`
- `notarius.Runner`
- `storage.ObjectStore`
- `notify.Sender`
- `analyzer.Runner`
Current execution usage:
- Actively used by implemented stages: `WhisperX`, `Seriatim`, `Audita`, `Scriptorium`, `ObjectStore`, `Notifier`.
- Present but not used by implemented stage set: `Analyzer`, legacy `storage.Backend`.
## Ownership
Default construction in app runner:
- Auto-constructed when not injected: WhisperX HTTP client, Seriatim subprocess runner, Audita subprocess runner, Scriptorium subprocess runner, object store (only when needed), and `notify.NoopSender`.
- Callers can inject test/fake implementations through `app.RunOptions.Env`.
Adapters own:
## State and manifest behavior
- Adapters do not directly mutate session/run manifests.
- Stages and runner own manifest writes and stage status transitions.
- Adapter outputs are persisted indirectly through stage result mapping (outputs/logs/generated configs/metadata).
- HTTP/subprocess/SDK argument and transport details.
- Backend-specific request/response mapping.
## Skip and resume behavior
- No adapter-level skip/resume semantics.
- Skip/resume/force behavior is decided by app runner using manifest stage state.
Adapters do not own:
## Failure behavior
- Adapter constructors validate config-derived values and fail early on invalid required inputs.
- Adapter run-time failures are returned to stage code with boundary context and are recorded as stage failures by runner logic.
- Subprocess adapters preserve stdout/stderr and generated-config paths to aid diagnosis.
- stage ordering/skip/force logic;
- manifest transitions;
- canonical path policy.
## Tests to inspect before changing
## Default Wiring
`internal/app/runner.go` initializes default adapters when not injected:
- WhisperX HTTP client from pipeline config.
- Seriatim subprocess runner.
- Audita subprocess runner.
- Scriptorium subprocess runner.
- Notarius subprocess runner when extraction is enabled.
- Noop notifier (`notify.NoopSender`).
- Object store only when required by selected stages/config.
Notarius is composed only when extraction is enabled; the extract stage owns
receipt, bundle, and configured-lane policy rather than the adapter.
Object-store construction goes through `newCommandObjectStore`, which loads
configured filesystem secrets before adapter initialization.
## Failure Semantics
- Constructor errors fail stage execution setup early.
- Runtime adapter errors propagate to stage code and then manifest failure handling.
- Subprocess adapters persist stage logs/generated configs through stage-managed paths.
## Implementation And Tests
- Composition: `internal/app/runner.go`, `internal/app/object_store.go`
- Shared subprocess mechanics: `internal/adapters/subprocess`
- Focused adapters: `internal/adapters/{whisperx,seriatim,audita,scriptorium,notarius,storage,notify}`
- `internal/adapters/whisperx/http_test.go`
- `internal/adapters/seriatim/subprocess_test.go`
- `internal/adapters/audita/subprocess_test.go`
- `internal/adapters/scriptorium/subprocess_test.go`
- `internal/adapters/notarius/subprocess_test.go`
- `internal/adapters/storage/*_test.go`
- `internal/adapters/notify/fake_test.go`
- `internal/adapters/analyzer/fake_test.go`
- `internal/app/runner_test.go`
## Architectural invariants
- Stage code depends on adapter interfaces, not transport-specific implementation types.
- External SDK-specific types remain inside adapter implementations.
- Default app wiring must remain deterministic and overrideable via injected env dependencies.
See the [WhisperX](../integrations/whisperx.md),
[Seriatim](../integrations/seriatim.md), [Audita](../integrations/audita.md),
[Scriptorium](../integrations/scriptorium.md), and
[Notarius](../integrations/notarius.md) contracts before changing an
externally visible boundary. Operator-selected values belong in
[Configuration](../config.md).

View File

@@ -1,87 +1,161 @@
# Internal: Artifacts
## Purpose
Define Narratio's artifact identity and resolution model for built-in transcript/bounds artifacts and runtime-configured analyze artifacts.
## Inputs and outputs
Inputs:
- artifact sources from config/runtime (`pipeline.scriptorium.artifacts.*.inputs.*.source`)
- session paths and optional session manifest stage outputs
- runtime artifact catalog state for configured artifact sources
Explain the artifact registry, runtime catalog, resolver, previous-input
requirements, and shared remote current-state mechanics implemented by
`internal/artifacts`. Configuration fields that accept source IDs belong in
[Configuration](../config.md); physical placement belongs in
[Operations](../operations.md).
Outputs:
- resolved local artifact path and provenance (`ResolvedSessionArtifact`)
- runtime catalog entries for planned/executable/available artifacts
- validation errors for unsupported, missing, or invalid artifact sources
## Built-in Source IDs
## Boundaries
Owns:
- built-in artifact registry and content validation rules
- runtime artifact catalog for configured artifact source IDs
- source resolution behavior for built-in and configured artifact sources
The internal registry recognizes these stable built-in source IDs:
Does not own:
- artifact generation (stages produce files)
- manifest transition policy
- archive promotion behavior
- `narratio.transcript.base`
- `narratio.transcript.polished`
- `narratio.transcript.final`
- `narratio.transcript.final_trimmed`
- `narratio.transcript.final_markdown`
- `narratio.transcript.final_trimmed_markdown`
- `narratio.bounds.session`
## Config fields used
- `pipeline.scriptorium.artifacts.<name>.enabled`
- `pipeline.scriptorium.artifacts.<name>.output_path`
- `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source`
Registry entries bind each ID to its producer, output kind, canonical fallback,
and content validator. The focused stage documents own their input/output flow;
[Configuration](../config.md) owns where operators may select these IDs.
## External adapters used
- none
## Configured, Extraction, And Previous-Session Sources
## State and manifest behavior
Built-in registry entries:
- configured source ID format: `narratio.artifact.<artifact_key>`
- extraction source ID format: `narratio.extraction.<output_key>`
- previous-session source ID format: `narratio.previous_session.artifact.<artifact_key>`
| Artifact ID | Canonical file | Producer stage | Output kind |
| --- | --- | --- | --- |
| `narratio.transcript.merged` | `transcripts/merged.json` | `merge` | `transcript_merged` |
| `narratio.transcript.polished` | `transcripts/processed.json` | `polish` | `transcript_processed` |
| `narratio.transcript.full` | `transcripts/normalized.json` | `normalize` | `transcript_normalized` |
| `narratio.transcript.trimmed` | `transcripts/trimmed.json` | `trim` | `transcript_trimmed` |
| `narratio.bounds.session` | `artifacts/session_bounds.json` | `trim` | `session_bounds` |
All formats are validated by strict source-policy rules. Extraction sources are
registered only from `pipeline.notarius.outputs`; the Notarius index has no
selectable source ID.
Runtime catalog entries include built-ins and configured `narratio.artifact.<name>` sources.
## Runtime Catalog
Catalog states:
- `planned`: source is registered and known for this run
- `executable`: configured artifact is selected for analyze execution
- `available`: artifact has a usable file path (generated this run or reused from disk)
`ArtifactCatalog` tracks:
Resolution behavior:
- built-in sources resolve via manifest producer outputs first, then canonical fallback path
- configured `narratio.artifact.<name>` sources resolve through runtime catalog availability
- configured source lookup requires catalog context
- `planned`: source registered for run context;
- `executable`: selected and enabled for analyze execution;
- `available`: local file exists and validates;
- `provenance`: availability source.
Current provenance values:
Configured artifact provenance values:
- `generated.current_analyze_run`
- `filesystem.disabled_artifact_output`
- `manifest.inputs.previous_cache`
- `current_session.previous_cache`
Content validation:
- transcript built-ins: JSON with top-level `segments` array
- bounds built-in: valid JSON
- configured artifacts: non-empty text file
## Resolution Rules
## Skip and resume behavior
- resolver and catalog have no direct skip/resume decisions
- stage/runner skip-resume behavior consumes catalog/resolver results
Built-ins:
## Failure behavior
- unsupported source -> source validation error
- known source unavailable -> `ErrSessionArtifactNotFound`
- configured source without catalog -> resolution error
- resolved file with invalid content -> validation error
1. manifest producer outputs (when present)
2. canonical session-path fallback
## Tests to inspect before changing
- `internal/artifacts/artifact_resolver_test.go`
- `internal/artifacts/catalog_test.go`
- `internal/stage/analyze_test.go`
- `internal/config/scriptorium_test.go`
Configured sources (`narratio.artifact.*`):
## Architectural invariants
- built-in IDs are static and registry-backed
- configured artifact IDs are runtime-derived (`narratio.artifact.<name>`) and catalog-backed
- built-in/source resolution remains deterministic and validation-gated
- resolve only through runtime catalog availability.
Extraction sources (`narratio.extraction.*`):
- use the shared registration and manifest hydration path in
`extraction_catalog.go`;
- require a current successful extract record with the exact configured source,
compatible contract and Notarius provenance, a confined regular durable
payload, and matching checksum; and
- are never inferred by scanning the Notarius bundle directory.
Previous-session sources (`narratio.previous_session.artifact.*`):
- resolve only from local `previous/` cache state;
- prefer manifest-backed previous-input paths;
- fallback to existing previous-cache filesystem paths.
Validation by content type:
- transcript JSON built-ins: JSON with top-level `segments` array;
- transcript Markdown built-ins: non-empty text file;
- bounds built-in: valid JSON;
- configured/previous-session artifact files: non-empty text file.
## Previous Requirement Collection
`CollectPreviousArtifactRequirements`:
- scans enabled configured artifacts only;
- extracts only canonical previous-session sources;
- deduplicates by artifact key;
- merges required and optional references (required wins);
- returns deterministic ordering and source locations.
## Current-State Helpers
Artifacts package owns shared remote current-state loading mechanics used by
restore, status and validation checks, and previous-cache planning.
Core helpers:
- `LoadCurrentRunPointer`
- `LoadCurrentManifest`
- `LoadCurrentState`
- `ValidateCurrentStateIdentity`
Typed missing-state errors:
- `CurrentRunPointerMissingError` (`ErrCurrentRunPointerMissing`)
- `CurrentManifestMissingError` (`ErrCurrentManifestMissing`)
Identity validation supports caller-provided expectations:
- expected campaign;
- expected session ID;
- expected run ID, or pointer/manifest run-ID consistency check.
Caller policy is intentionally outside artifacts helpers:
- some callers fail on missing current state;
- some callers downgrade missing state to status/findings;
- some callers skip optional behavior when state is missing.
## Key Path Helpers
`internal/artifacts/paths.go` and S3-key helpers define canonical helpers for:
- session/work/run paths;
- previous-cache paths;
- spool/cache paths;
- S3 session/run/current-state key layout.
See [Workspace Internals](workspace.md) for how callers consume local helpers
and [Operations](../operations.md#local-state-layout) for the authoritative
physical layout.
## Invariants
- source ID formats are stable contracts;
- artifact resolution is deterministic and manifest-aware;
- extraction sources are available only from a compatible successful manifest
record;
- previous-session source resolution in `analyze` is local-only;
- remote current-state key construction remains centralized in artifacts helpers.
## Implementation And Tests
- Registry and resolution: `internal/artifacts/artifact_resolver.go`,
`internal/artifacts/catalog.go`, `internal/artifacts/transcripts.go`,
`internal/artifacts/extraction_catalog.go`
- Current state: `internal/artifacts/current_state.go`
- Paths and keys: `internal/artifacts/paths.go`,
`internal/artifacts/s3_keys.go`
- Previous requirements: `internal/artifacts/previous_requirements.go`
- Tests: `internal/artifacts/artifact_resolver_test.go`,
`internal/artifacts/catalog_test.go`,
`internal/artifacts/extraction_catalog_test.go`,
`internal/artifacts/current_state_test.go`,
`internal/artifacts/paths_model_test.go`,
`internal/artifacts/previous_requirements_test.go`

View File

@@ -1,86 +1,81 @@
# Internal: Command Restore
## Purpose
Define the implemented `narratio restore` command contract: committed remote-state discovery, deterministic planning, safe file installation, conflict policy, and restore reporting.
## Inputs and outputs
Inputs:
- CLI flags: `--config`, `--session`, `--session-id`, `--dry-run`, `--force`, `--include-audio`.
- Resolved/validated `pipeline.yml` and `session.yml`.
- Configured remote object store.
- Remote committed current-state markers (`current/run_id.txt`, `current/manifest.json`).
Explain the implemented restore discovery, planning, installation, and
reporting flow in `internal/app`. User invocation belongs in
[CLI](../cli.md#session-restore), and the operator recovery procedure and
physical restore scope belong in
[Operations](../operations.md#restore-workflow).
Outputs:
- Dry-run summary to stdout (plan + counts).
- Non-dry-run completion summary to stdout.
- Local durable session files restored under canonical session root.
- Non-dry-run restore report at `reports/restore-latest.json`.
Restore is split into explicit phases so remote authority, local conflict
policy, and filesystem mutation can be tested independently.
## Boundaries
Owns:
- Restore command flag parsing and command wiring.
- Remote current-state discovery and identity validation.
- Restore plan construction and conflict classification.
- Restore execution for planned downloads.
- Restore report model and persistence.
## Discovery Contract
Does not own:
- Stage execution orchestration (`run`, `resume`, `run-stage`).
- Archive publish behavior (owned by archive stage).
- Storage transport implementation details (owned by storage adapters).
Discovery delegates current-state pointer and manifest loading to
`internal/artifacts`, then validates the result against the resolved request:
## Config fields used
- Config/session discovery and templating fields consumed by all commands.
- `pipeline.workspace.root` (local restore target root).
- `pipeline.storage.*` (remote backend + archive identity derivation).
- `pipeline.storage.s3.*` identity components used by archive prefix helpers.
- `session.session_id`
- `session.campaign`
- campaign must match;
- session ID must match.
## External adapters used
- `storage.ObjectStore` for `Exists`, `List`, `Download`.
- `artifacts.Store` (`LocalStore`) for layout and session lock management.
- `manifest.LocalStore` for manifest decode/validation and identity checks.
Restore treats any missing or invalid remote current state as a command error.
## State and manifest behavior
- Restore is not a pipeline run and does not create a run manifest.
- Restore uses committed remote current state only:
- `current/run_id.txt` must exist and be non-empty.
- `current/manifest.json` must decode and match requested session/campaign.
- Non-dry-run writes restore files to canonical session paths.
- Manifest install behavior:
- validated before replacement.
- installed last among download actions.
- existing local manifest is preserved if restored manifest validation/install fails.
- Non-dry-run report persists summary/action status metadata in `reports/restore-latest.json`.
## Planning Contract
## Skip and resume behavior
- Restore does not participate in stage skip/resume decisions.
- Restore provides durable local state so subsequent stage commands can resume or rerun based on restored manifest state.
- Dry-run is read-only and returns plan output only.
Restore planner action kinds:
## Failure behavior
- Fails when storage backend is unavailable or archive identity cannot be resolved.
- Fails when remote current pointer/manifest is missing or invalid.
- Fails when remote manifest identity mismatches requested campaign/session.
- Fails on local conflicts unless `--force` is set.
- Fails fast on session lock acquisition conflict for non-dry-run execution.
- On execution failure, previously installed files remain; no rollback is performed.
- `download`;
- `skip_same`;
- `conflict`.
## Tests to inspect before changing
- `internal/app/restore_test.go`
- `internal/app/restore_discovery_test.go`
- `internal/app/restore_plan_test.go`
- `internal/app/restore_execution_test.go`
- `internal/app/restore_workflow_test.go`
- `internal/artifacts/archive_identity_test.go`
Planner behavior:
## Architectural invariants
- Restore relies on centralized archive identity/key helpers (`internal/artifacts`) rather than ad hoc key building.
- `current/run_id.txt` is the remote commit marker; restore must not infer committed state from incidental files.
- Local path mapping is traversal-safe and constrained to session root.
- Restore scope is deterministic and path-classified:
- include `manifest.json`, `transcripts/**`, `artifacts/**`
- include `audio/**` only with `--include-audio`
- exclude `runs/**`, `logs/**`, `reports/**`, `config/**`, `inputs/**`
- Command remains standalone; no implicit `run --restore` behavior.
- remote list scope is the resolved session prefix;
- remote-to-local mapping is traversal-safe;
- actions are sorted by local relative path and then remote key;
- force converts differing local targets from conflicts to downloads.
Previous-cache files are planned separately through `previouscache.BuildPlan`
when configured previous-session requirements exist.
## Execution Contract
Execution order and safety:
- non-manifest downloads happen before manifest install;
- `manifest.json` installs last;
- downloads use sibling temp files plus atomic rename;
- manifest replacement is validated before rename;
- failed installs do not roll back files already written in the same execution.
Audio restore path:
- uses `audio.MaterializeS3Audio`;
- integrates spool and S3 audio cache paths;
- supports cache-hit reuse without object redownload.
## Reporting Contract
- dry-run mode prints a summary and performs no local writes;
- execution mode persists the canonical restore report described in
[Operations](../operations.md#restore-workflow);
- report includes plan counts, per-action status, and execution failures.
## Invariants
- restore uses committed remote current state as authority;
- `current/run_id.txt` is the remote publish commit marker;
- restore does not execute pipeline stages.
## Implementation And Tests
- Discovery: `internal/app/restore_discovery.go`
- Planning: `internal/app/restore_plan.go`, `internal/previouscache`
- Execution: `internal/app/restore_execute.go`
- Reporting and command coordination: `internal/app/restore_report.go`,
`internal/app/restore.go`
- Tests: `internal/app/restore_discovery_test.go`,
`internal/app/restore_plan_test.go`,
`internal/app/restore_execution_test.go`,
`internal/app/restore_workflow_test.go`

View File

@@ -1,81 +1,100 @@
# Internal: Manifest
## Purpose
Describe Narratio's durable execution state model for session-level and run-level manifests, including lifecycle transitions and persistence behavior.
## Inputs and outputs
Inputs:
- Session identity and run identity from app orchestration.
- Stage transition events and stage result payloads.
Explain the session-progress and invocation-audit models implemented by
`internal/manifest`. Physical manifest placement belongs in
[Operations](../operations.md#local-state-layout).
Outputs:
- Session manifest at `{workspace.root}/work/{campaign}/{session_id}/manifest.json`.
- Run manifest at `{workspace.root}/work/{campaign}/{session_id}/runs/{run_id}/manifest.json`.
## Session Manifest
## Boundaries
Owns:
- Manifest schemas (`Manifest`, `RunManifest`, stage records, error records, input/artifact records).
- Stage status/action transition methods.
- Persistent store contract (`manifest.Store`) and local JSON store implementation.
`manifest.Manifest` records:
Does not own:
- Stage implementation details.
- Path construction policy outside manifest file persistence calls.
- CLI command behavior.
- identity (`session_id`, `campaign`, `run_id`)
- local path metadata (`local_workdir`, `local_spool_dir`)
- remote identity metadata (`s3_bucket`, `s3_session_prefix`, `s3_run_prefix`)
- `inputs` records
- durable `artifacts` records
- per-stage `stages` map
## Config fields used
Manifest package itself does not read config directly.
The model admits these stage states:
Manifest identity fields are populated by app/stage orchestration from:
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.storage.s3.*` (when archive/S3 identity is set)
- `pending`
- `running`
- `succeeded`
- `failed`
- `skipped`
- `stale`
- `interrupted`
## External adapters used
- No external service adapters.
- Uses local filesystem for persistence via `manifest.LocalStore`.
## Run Manifest
## State and manifest behavior
Session manifest model:
- Tracks durable per-session stage state and provenance (`pending`, `running`, `succeeded`, `failed`, `skipped`, `stale`, `interrupted`).
- Stores resolved inputs, durable artifacts, stage logs/config refs, and stage metadata.
`manifest.RunManifest` is created for each invocation and records:
Run manifest model:
- Tracks one invocation (`run_id`) with requested stages and force mode.
- Tracks per-stage action (`run` or `skip`) and per-stage status.
- Tracks overall run status (`running`, `succeeded`, `failed`).
- invocation identity and `force` flag
- requested stages
- per-stage action (`run` or `skip`)
- per-stage status
- overall run status (`running`, `succeeded`, `failed`)
Persistence behavior:
- Load validates required identity/timestamp fields and normalizes maps/records.
- Save updates `updated_at` and writes JSON atomically (temp file + rename).
- Session and run manifests are saved incrementally before/after stage transitions.
## Persistence Semantics
Relationship during execution:
- Runner updates both manifests for every stage transition.
- Session manifest is the durable pipeline-progress ledger.
- Run manifest is invocation history and audit record.
- Analyze stage outputs are persisted as `kind=scriptorium_artifact` with `source_id=narratio.artifact.<name>` for configured artifact identity.
`manifest.LocalStore`:
## Skip and resume behavior
- Resume and skip decisions are based on session-manifest stage statuses.
- `--force` reruns selected stages and marks downstream succeeded stages as `stale` in session manifest.
- Run manifest records whether each stage was executed or skipped in that invocation.
- validates loaded documents;
- normalizes missing maps/stage records;
- writes atomically via temp file + rename;
- updates `updated_at` on save.
## Failure behavior
- Stage failure marks both manifests failed for that stage and records error messages/timestamps.
- Save failures are returned immediately and fail the command.
- Invalid/malformed manifest files fail load with explicit validation/decode errors.
## Execution Semantics
## Tests to inspect before changing
- `internal/manifest/manifest_test.go`
- `internal/manifest/run_manifest_test.go`
- `internal/manifest/store_test.go`
- `internal/app/runner_test.go`
- `internal/app/run_control_test.go`
- `internal/app/resume_run_stage_test.go`
The application runner marks an executing stage running and then succeeded or
failed in both manifests, persisting each transition. On success it records
outputs, logs, generated configuration references, and metadata. Artifact
records may include optional contract and external provenance objects; old
manifests remain compatible when those fields are absent. A successful forced
rerun marks only succeeded downstream session-stage records stale.
## Architectural invariants
- Session manifest is authoritative for stage progression across invocations.
- Run manifest is invocation-scoped and never replaces session manifest as progress authority.
- Manifest writes are atomic and deterministic (JSON + newline, temp rename pattern).
Starting an execution clears the current session-stage record's prior outputs,
logs, generated configuration references, and metadata. Failed and skipped
transitions enforce the same clearing rule directly, while success repopulates
only fields returned by the new result. Marking a record stale does not clear
those details because resume validation and diagnosis may still require them
before execution begins. Invocation run manifests remain immutable audit
records of their own outcomes.
A stage may explicitly return a skipped disposition and stable reason. The
runner persists that outcome in both manifests, clears older outputs for the
session-stage record along with older logs, generated configuration references,
and metadata, then applies any bounded details from the current skip and
continues. This self-skip is distinct from deciding not to execute an
already-succeeded stage and is reconsidered on later runs. Skipped results
cannot contain outputs.
When an already-succeeded stage is skipped, the invocation run manifest records
the `skip` action and reason. The session manifest deliberately retains its
existing succeeded record because it remains the cross-invocation progress
authority. Stages with a resume validator, currently extraction, may reject an
otherwise eligible skip when the recorded durable result is obsolete; the
runner marks it stale and executes it.
Session manifest is the authoritative stage-progress ledger across invocations.
Run manifest is invocation-scoped audit state.
## Invariants
- stage resume/skip decisions are session-manifest driven.
- running, failed, and self-skipped stages do not retain result payloads from
an earlier success.
- stale stages retain prior details until replacement execution starts.
- force reruns stale downstream succeeded stages.
- run manifest does not replace session manifest as progress authority.
## Implementation And Tests
- Models and transitions: `internal/manifest/manifest.go`,
`internal/manifest/run_manifest.go`
- Persistence and validation: `internal/manifest/store.go`
- Package tests: `internal/manifest/*_test.go`
- Assembled execution behavior: `internal/app/runner_test.go`,
`internal/app/run_stage_test.go`

98
docs/internal/overview.md Normal file
View File

@@ -0,0 +1,98 @@
# Internal Overview
This document is the implemented component map for Narratio. Normative system
boundaries and dependency direction belong in
[Architecture](../policy/architecture.md). User and operator contracts belong
in the [CLI](../cli.md), [Configuration](../config.md),
[Operations](../operations.md), and [Troubleshooting](../troubleshooting.md).
Externally observable tool and format contracts belong under
[Integrations](../integrations/).
## Execution Path
```text
cmd/narratio -> internal/app -> configuration and production composition
-> internal/stage -> adapters and external systems
-> manifests and artifact resolution -> durable local/remote output
```
The executable delegates process behavior to the application boundary. The
application resolves configuration, composes concrete collaborators, acquires
session safety controls, and runs commands. Pipeline commands execute the
canonical stage sequence through adapter interfaces, while manifests record
progress and artifact services resolve durable inputs and outputs.
## Components
| Area | Implemented owners | Responsibility |
| --- | --- | --- |
| Executable | `cmd/narratio` | Process entry, standard stream wiring, argument handoff, and exit status. |
| Application orchestration | `internal/app` | Command dispatch, configuration selection, secret-file environment loading, production composition, session locking, planning, execution, restore, cleanup gates, and user-facing reporting. |
| Configuration | `internal/config` | Strict YAML loading, discovery, defaults, normalization, session templating, and validation. |
| Pipeline stages | `internal/stage` | Canonical stage registry, shared stage contract, execution dependencies, and implemented stage behavior. |
| External boundaries | `internal/adapters`, `internal/audio` | WhisperX HTTP, downstream subprocesses, notification, object storage, and S3 audio materialization behind Narratio contracts. |
| Manifests | `internal/manifest` | Durable session progress, invocation audit state, stage transitions, validation, and atomic persistence. |
| Artifacts and paths | `internal/artifacts`, `internal/pathsafe` | Artifact identities and resolution, local and remote path/key models, current-state discovery, and confined relative destinations. |
| Previous-session cache | `internal/previouscache` | Deterministic planning and materialization requirements for configured previous-session inputs. |
| Artifact policy | `internal/artifactpolicy` | Source and destination policy, configured artifact identity validation, and publish destination safety. |
| Shared models and file operations | `internal/artifactmodel`, `internal/contracts`, `internal/fileops` | Transcript and artifact data contracts plus narrow atomic filesystem helpers. |
| Logging | `internal/logging` | Application logger construction and shared structured logging behavior. |
The application boundary composes concrete implementations. Stages depend on
Narratio-level contracts; external transport and SDK details remain in
adapters. The normative rules for these relationships remain in
[Architecture](../policy/architecture.md).
## Pipeline Stage Set
The implemented canonical order is:
1. [`prepare`](stage-prepare.md)
2. [`transcribe`](stage-transcribe.md)
3. [`merge`](stage-merge.md)
4. [`polish`](stage-polish.md)
5. [`normalize`](stage-normalize.md)
6. [`trim`](stage-trim.md)
7. [`extract`](stage-extract.md)
8. [`render`](stage-render.md)
9. [`analyze`](stage-analyze.md)
10. [`publish`](stage-publish.md)
11. `notify` (placeholder)
`notify` currently has optional notifier call behavior and no persisted pipeline
outputs; its default collaborator is a no-op sender. The focused stage
documents own implementation mechanics. The
[CLI](../cli.md) and [Operations](../operations.md) own user-visible invocation
and execution semantics.
## Focused Documentation
- [Adapter Internals](adapters.md): external adapter boundaries, composition,
failure behavior, and test surfaces.
- [Artifact Internals](artifacts.md): source identities, runtime catalog,
resolution, previous requirements, and current-state helpers.
- [Manifest Internals](manifest.md): session and run records, persistence, and
execution transitions.
- [Storage Internals](storage.md): object-store interface and S3 behavior.
- [Workspace Internals](workspace.md): local layout, locking, and cleanup
guardrails.
- [Restore Internals](command-restore.md): discovery, planning, execution, and
reporting.
- [`prepare`](stage-prepare.md)
- [`transcribe`](stage-transcribe.md)
- [`merge`](stage-merge.md)
- [`polish`](stage-polish.md)
- [`normalize`](stage-normalize.md)
- [`trim`](stage-trim.md)
- [`extract`](stage-extract.md)
- [`render`](stage-render.md)
- [`analyze`](stage-analyze.md)
- [`publish`](stage-publish.md)
Use this map to find an owner, then read its focused documentation and tests
before changing behavior.
The stage registry is implemented in `internal/stage/placeholders.go` and its
ordering is protected by `internal/app/planner_test.go`. Cross-invocation skip,
force, failure, and invalidation behavior is exercised in
`internal/app/runner_test.go` and `internal/app/run_stage_test.go`.

View File

@@ -1,84 +1,60 @@
# Stage: analyze
## Purpose
Execute selected configured Scriptorium artifacts in deterministic dependency order and promote successful outputs to canonical session artifact paths.
## Inputs and Outputs
Inputs:
- configured artifact definitions from `pipeline.scriptorium.artifacts`
- selected artifact filter from runtime (`--artifacts`) when provided
- resolved artifact input sources declared per artifact (`inputs.*.source`)
- optional previous-session file inputs (`previous_session_artifact`)
Execute selected configured Scriptorium artifacts in dependency order and materialize outputs.
Outputs:
- one promoted output file per executed configured artifact at that artifact's configured `output_path`
- stage metadata containing generated artifact entries and reused disabled-artifact entries
## Inputs
## Boundaries
Owns:
- runtime artifact catalog construction for analyze execution
- selected-artifact planning and dependency ordering
- per-artifact input resolution, var resolution, timeout/render-debug resolution
- Scriptorium run/render invocation for each selected artifact
- run-local output generation and canonical promotion
- configured artifacts from `pipeline.scriptorium.artifacts`
- optional selected artifact keys supplied through the stage environment
- built-in/configured/previous-session source references in artifact inputs
Does not own:
- transcript generation/processing stages
- archive promotion policy
- per-artifact resume semantics
Supported source families:
- built-ins: `narratio.transcript.*`, `narratio.bounds.session`
- prepared stable inputs: `narratio.input.players`, `narratio.input.party`,
`narratio.input.glossary`
- configured artifacts: `narratio.artifact.<key>`
- previous-session cache: `narratio.previous_session.artifact.<key>`
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.scriptorium.binary`
- `pipeline.scriptorium.config_path`
- `pipeline.scriptorium.timeout`
- `pipeline.scriptorium.render_debug`
- `pipeline.scriptorium.artifacts.<name>.*`
- `enabled`
- `depends_on`
- `prompt_id`
- `profile_id`
- `timeout`
- `output_path`
- `render_debug`
- `inputs`
- `vars`
## Outputs
## External Adapters Used
- Scriptorium adapter:
- optional `RenderArtifact` (render debug)
- `RunArtifact` (artifact generation)
- one materialized output per executed configured artifact (`output_path`)
- stage metadata describing selected/generated/reused artifacts
## State and Manifest Behavior
- If `pipeline.scriptorium` is absent, stage returns success metadata with `skipped=true`.
- If no artifacts are configured, stage returns success metadata with `skipped=true`.
- If zero artifacts are executable after `enabled` + `--artifacts` filtering, stage returns success metadata with `skipped=true`.
- Builds runtime catalog with built-ins and configured artifacts.
- Non-executable configured artifacts are marked available only when their configured output file exists and is valid on disk.
- Executes selected configured artifacts in topological order with deterministic tie-breaking.
- For each generated artifact, records metadata fields including `name`, `source_id`, `output_kind`, `path`, `prompt_id`, `profile_id`, and `provenance`.
- Reused disabled artifacts are recorded separately in `reused_artifacts` with provenance `filesystem.disabled_artifact_output`.
## Key Behavior
## Skip and Resume Behavior
- Runner-level skip applies when analyze is already `succeeded` and `--force` is not set.
- Analyze remains stage-scoped for resume/skip; there is no per-artifact resume state.
- `--artifacts` filters which configured artifacts are executable when analyze runs; it does not imply `--force`.
- skips with metadata when Scriptorium config is missing or no executable artifacts remain.
- builds runtime artifact catalog (built-ins + configured artifacts).
- marks non-executable configured artifacts as reusable when output files already exist.
- validates selected artifact dependency order (cycle-safe topo ordering).
- resolves required/optional inputs per artifact source definition.
- resolves prepared stable input sources from `inputs/*.yml` materialized by `prepare`.
- resolves previous-session sources from local `previous/` cache only.
- runs optional render-debug, then artifact execution.
- validates non-empty output files and materializes canonical outputs.
## Failure Behavior
- Fails on invalid dependency ordering, unavailable required configured inputs, invalid built-in input prerequisites, render/run adapter failures, validation-failed adapter results, or missing/empty outputs.
- Required configured dependency missing from catalog availability fails clearly before invocation.
- Optional missing inputs are omitted.
## Failure Semantics
## Tests to Inspect Before Changing
- `internal/stage/analyze_test.go`
- `internal/artifacts/catalog_test.go`
- `internal/artifacts/artifact_resolver_test.go`
- `internal/adapters/scriptorium/subprocess_test.go`
- required missing configured/previous-session inputs fail.
- missing required prepared stable input source includes prepare rerun guidance.
- missing required previous-session source includes prepare rerun guidance.
- missing required `narratio.transcript.final_markdown` or
`narratio.transcript.final_trimmed_markdown` inputs includes render rerun
guidance.
- dependency cycles or unavailable required dependencies fail.
- adapter validation failures fail stage.
## Architectural Invariants
- Configured artifacts are identified by `narratio.artifact.<name>` source IDs.
- Artifact-to-artifact references rely on explicit `depends_on` declarations validated in config.
- Generated analyze outputs are treated uniformly as Scriptorium artifacts.
- Successful outputs must exist and be non-empty before promotion.
## Invariants
- `analyze` performs no remote storage calls for previous-session source resolution.
- output provenance and metadata are deterministic per execution.
## Related Contracts And Tests
- [Configuration](../config.md#scriptorium-artifact-entries) owns artifact
fields and source-selection rules.
- [CLI](../cli.md) owns user-visible artifact selection.
- [Scriptorium](../integrations/scriptorium.md) owns the subprocess contract.
- Implementation and tests: `internal/stage/analyze.go`,
`internal/stage/analyze_test.go`

View File

@@ -1,68 +0,0 @@
# Stage: archive
## Purpose
Publish run records and promoted session artifacts to object storage, then atomically advance the remote current pointer.
## Inputs and Outputs
Inputs:
- session manifest and prerequisite stage records
- run root contents under `runs/{run_id}/`
- promotion rules with artifact `source` IDs and archive `dest` paths (`archive.promote_artifacts`)
Outputs:
- uploaded run files under `{session_prefix}/runs/{run_id}/...`
- uploaded promoted artifacts under `{session_prefix}/...`
- `{session_prefix}/current/manifest.json`
- `{session_prefix}/current/run_id.txt` written last
## Boundaries
Owns:
- Archive enable/disable gate behavior
- Prerequisite stage success enforcement
- Run file collection and upload (excluding `audio/`)
- Promotion rule resolution and upload
- Commit pointer publish order
Does not own:
- Stage execution before archive
- Post-archive local cleanup policy execution (handled by app cleanup logic)
## Config Fields Used
- `pipeline.archive.enabled`
- `pipeline.archive.upload_run`
- `pipeline.archive.promote_artifacts`
- `pipeline.storage.s3.bucket`
- `pipeline.storage.s3.root_prefix`
- `pipeline.workspace.root`
- `session.campaign`
- `session.session_id`
## External Adapters Used
- Object storage backend (`env.ObjectStore`) for upload/list primitives.
## State and Manifest Behavior
- Requires `prepare`, `transcribe`, `merge`, `polish`, `normalize`, `trim`, and `analyze` status `succeeded`.
- Resolves bucket/prefix from manifest identity first, then config fallback.
- Writes metadata including:
- upload counts/paths
- `current_manifest_key`
- `current_run_id_key`
- `current_pointer_written`
- On skipped archive path, returns metadata with `skipped=true` and pointer not written.
## Skip and Resume Behavior
- Stage may self-skip (metadata skip) when archive disabled or run upload disabled.
- Runner-level skip also applies for previously succeeded stage unless forced.
## Failure Behavior
- Fails on missing prerequisite success, missing object store when required, missing run root, missing required promotion source, upload failures, or pointer write failures.
- Pointer semantics are fail-safe: `current/run_id.txt` is not written if prior required uploads fail.
## Tests to Inspect Before Changing
- `internal/stage/archive_test.go`
- `internal/app/post_archive_cleanup_test.go`
## Architectural Invariants
- Run upload excludes `audio/` subtree.
- `current/manifest.json` uploads before `current/run_id.txt`.
- `current/run_id.txt` is the remote publish commit marker.

View File

@@ -0,0 +1,84 @@
# Internal: Extract Stage
## Responsibility
`extract` runs after `trim` and before `render`. It converts the canonical
`narratio.transcript.final_trimmed` JSON into configured Notarius lane artifacts.
An omitted or disabled Notarius section makes the stage explicitly self-skip
with reason `notarius_disabled`, no outputs, and no Notarius runner.
The external protocol is documented in the
[Notarius integration contract](../integrations/notarius.md). Configuration
fields belong in [Configuration](../config.md), and physical paths and force
procedures belong in [Operations](../operations.md).
## Lifecycle
`internal/stage/extract.go`:
1. resolves the final trimmed transcript from the shared artifact catalog;
2. resolves and fingerprints the Notarius invocation contract;
3. creates a run-local staging directory and invokes the injected
`notarius.Runner`;
4. validates the successful receipt, confined index, configured required lane
descriptors, and regular payload files;
5. atomically promotes the complete bundle to its immutable durable location;
6. records one non-selectable `notarius_index` output and one selectable
`notarius_lane` output per configured lane; and
7. registers each lane as `narratio.extraction.<output_key>` for downstream
Scriptorium and publish resolution.
Lane records retain checksum, contract, producer run ID, and Notarius system,
run, pipeline, and lane provenance. Stage metadata retains the durable bundle
root, receipt, diagnostic paths, rejection/warning summaries, producing
Narratio run ID, and invocation fingerprint. Validation completes before
promotion, so a rejected result cannot expose a partial durable bundle.
Any executed extraction outcome that replaces a different effective outcome
marks succeeded downstream stages stale. Repeating the same disabled self-skip
with no outputs is stable and does not repeatedly invalidate downstream stages.
## Resume Validation
`internal/stage/extract_resume.go` permits a skip only when the existing stage
record succeeded and still matches the current invocation fingerprint. The
fingerprint covers the resolved executable and config paths, pipeline ID,
timeout, working directory, and sorted configured output contracts.
The validator then checks the producing run identity, canonical immutable
bundle root, path confinement and absence of symlink components, receipt
identity, exactly one canonical index, the exact configured source set,
contracts and provenance, regular-file status, and stored checksums. Missing or
obsolete results are non-resumable and run again; unsafe filesystem conditions
return an error rather than silently accepting or replacing data.
The fingerprint cannot observe files imported by Notarius configuration,
profile contents, prompt/module definitions, or other transitive inputs.
Operators must force extraction after changing any such input.
## Failure Behavior
Adapter startup, timeout, nonzero exit, receipt decoding, path confinement,
index compatibility, required-lane rejection, payload inspection, checksum, or
promotion errors fail the stage through ordinary manifest transition handling.
Stdout receipt and stderr diagnostics remain separate. Downstream stages are
not given selectable extraction sources unless the complete configured result
has passed validation and promotion.
When a replacement attempt begins, the current session-stage record no longer
advertises payload from the previous success. A failed replacement therefore
has no current outputs, logs, generated configuration references, or metadata,
while the earlier invocation manifest and immutable promoted bundle remain
available for audit and recovery.
## Implementation And Focused Tests
- Stage execution, selection, and resume validation: `internal/stage/extract.go`,
`internal/stage/extract_resume.go`,
`internal/stage/extract_test.go`
- Subprocess boundary: `internal/adapters/notarius/subprocess.go`,
`internal/adapters/notarius/subprocess_test.go`
- Catalog hydration: `internal/artifacts/extraction_catalog.go`,
`internal/artifacts/extraction_catalog_test.go`
- Composition and downstream behavior: `internal/app/runner_test.go`,
`internal/stage/analyze_test.go`, `internal/stage/publish_test.go`

View File

@@ -1,63 +1,37 @@
# Stage: merge
## Purpose
Normalize per-speaker raw transcripts and merge them into one merged transcript via Seriatim.
## Inputs and Outputs
Inputs:
Normalize raw transcript inputs and merge into base transcript via Seriatim.
## Inputs
- `transcripts/raw/*.json`
- `inputs/speakers.yml`
- `inputs/autocorrect.yml`
Outputs:
- `transcripts/merged.json`
- optional `artifacts/seriatim.report.json` (when report enabled)
## Outputs
## Boundaries
Owns:
- Raw transcript discovery/validation
- Per-input normalize calls to Seriatim
- Final merge call to Seriatim
- Run-local log/config/report path wiring
- Promotion of merged/report outputs to canonical paths
- `transcripts/base.json`
- optional `artifacts/seriatim.report.json`
Does not own:
- Transcript polishing or downstream artifact generation
## Key Behavior
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.seriatim.binary`
- `pipeline.seriatim.timeout`
- `pipeline.seriatim.output_schema`
- `pipeline.seriatim.coalesce_gap`
- `pipeline.seriatim.report`
- `pipeline.seriatim.env.*`
- discovers and validates raw transcript inputs.
- normalizes each raw transcript (`seriatim.Normalize`) into run-local scratch output.
- merges normalized inputs (`seriatim.Run`) into base transcript.
- validates merged transcript and optional report JSON.
- materializes canonical outputs and records stage logs/generated configs.
## External Adapters Used
- Seriatim adapter:
- `Normalize` for each raw input
- `Run` for final merge
## Invariants
## State and Manifest Behavior
- Reads transcript inputs from transcribe stage outputs in manifest when present; falls back to canonical raw directory.
- Writes run-local outputs/logs/config under `runs/{run_id}/merge/...` when enabled.
- Promotes canonical merged transcript and optional report.
- Records normalized-input provenance and adapter metadata in stage metadata.
- merge always consumes normalized forms of raw inputs.
- base transcript must validate before stage success.
- report output is config-gated.
## Skip and Resume Behavior
- Runner-level skip applies when already succeeded and not forced.
- Forced rerun of this or upstream stages can stale downstream succeeded stages via runner invalidation.
## Related Contracts And Tests
## Failure Behavior
- Fails on missing/invalid raw transcripts, missing speakers/autocorrect files, normalize failure, merge failure, invalid merged output JSON, or invalid report JSON when enabled.
## Tests to Inspect Before Changing
- `internal/stage/merge_test.go`
- `internal/adapters/seriatim/subprocess_test.go`
## Architectural Invariants
- Merge consumes normalized forms of each raw transcript.
- Merged transcript must validate before promotion.
- Report output is optional and gated by config.
- [Seriatim](../integrations/seriatim.md) owns subprocess and output semantics.
- [Configuration](../config.md#pipeline) owns operator-selected Seriatim values.
- Implementation and tests: `internal/stage/merge.go`,
`internal/stage/merge_test.go`

View File

@@ -1,56 +1,34 @@
# Stage: normalize
## Purpose
Normalize the processed transcript into a deterministic intermediate schema for trim and optionally emit a normalize report.
## Inputs and Outputs
Inputs:
- `transcripts/processed.json`
Normalize polished transcript into final transcript using Seriatim.
Outputs:
- `transcripts/normalized.json` (or configured normalize output path)
## Inputs
- `transcripts/polished.json`
## Outputs
- `transcripts/final.json` (or configured normalize output path)
- optional `artifacts/seriatim.normalize.report.json`
## Boundaries
Owns:
- Processed transcript discovery/validation
- Normalize request construction and invocation
- Optional normalize report wiring
- Promotion of normalized transcript and optional report
## Key Behavior
Does not own:
- Bounds detection or segment trimming
- resolves polished transcript from manifest outputs/canonical fallback.
- applies `pipeline.normalize` config or default normalize config.
- runs Seriatim normalize with configured timeout/binary.
- validates normalized transcript and optional report.
- materializes canonical outputs and records logs/generated configs.
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.normalize.output_path`
- `pipeline.normalize.output_schema`
- `pipeline.normalize.report`
- `pipeline.seriatim.binary`
- `pipeline.seriatim.timeout`
## Invariants
## External Adapters Used
- Seriatim adapter (`Normalize`).
- final transcript must validate as processed transcript JSON (`segments` array).
- normalize defaults are applied when `pipeline.normalize` is unset.
## State and Manifest Behavior
- Reads processed transcript from polish outputs in manifest when present; falls back to canonical path.
- Uses run-local output/report/log/config paths when run layout is enabled.
- Promotes canonical normalized transcript and optional normalize report.
- Records adapter/result metadata including source path selection.
## Related Contracts And Tests
## Skip and Resume Behavior
- Runner-level skip applies when already succeeded and not forced.
- Forced reruns can stale downstream succeeded stages.
## Failure Behavior
- Fails on missing/invalid processed transcript, adapter error, invalid normalized output, or invalid report output when report enabled.
## Tests to Inspect Before Changing
- `internal/stage/normalize_test.go`
- `internal/adapters/seriatim/subprocess_test.go`
## Architectural Invariants
- Normalized output must validate as processed-transcript-compatible JSON (`segments` array required).
- Default normalize config is applied when `pipeline.normalize` is unset.
- [Seriatim](../integrations/seriatim.md) owns subprocess and output semantics.
- [Configuration](../config.md#pipeline) owns normalize fields and defaults.
- Implementation and tests: `internal/stage/normalize.go`,
`internal/stage/normalize_test.go`

View File

@@ -1,69 +1,36 @@
# Stage: polish
## Purpose
Polish merged transcript with Audita and produce a processed transcript for downstream normalization/analyze.
## Inputs and Outputs
Inputs:
- `transcripts/merged.json`
Run Audita polishing on base transcript and produce polished transcript.
## Inputs
- `transcripts/base.json`
- `inputs/glossary.yml`
Outputs:
- `transcripts/processed.json`
- optional `artifacts/audita.report.json` (when report enabled)
## Outputs
## Boundaries
Owns:
- Merged transcript discovery/validation
- Audita invocation request construction
- Run-local logs/config/work-dir/report wiring
- Promotion of processed transcript and optional report
- `transcripts/polished.json`
- optional `artifacts/audita.report.json`
Does not own:
- Upstream merge normalization
- Downstream normalize/trim/analyze logic
## Key Behavior
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.audita.binary`
- `pipeline.audita.timeout`
- `pipeline.audita.llm_api_key_env`
- `pipeline.audita.modules`
- `pipeline.audita.base_url`
- `pipeline.audita.model`
- `pipeline.audita.transcript_description`
- `pipeline.audita.config_path`
- `pipeline.audita.output_schema`
- `pipeline.audita.work_dir_retention`
- `pipeline.audita.total_llm_concurrency`
- `pipeline.audita.proposal_llm_concurrency`
- `pipeline.audita.validation_model`
- `pipeline.audita.validation_llm_concurrency`
- `pipeline.audita.report`
- resolves base transcript from merge outputs/canonical fallback.
- invokes Audita with configured model/module/runtime options.
- validates processed transcript structure (`segments` array required).
- validates optional report JSON.
- materializes canonical outputs; records logs/generated config and adapter metadata.
## External Adapters Used
- Audita adapter (`env.Audita.Run`).
## Invariants
## State and Manifest Behavior
- Reads merged transcript from merge manifest outputs when available; falls back to canonical merged path.
- Uses run-local output/report/log/config/scratch paths when run layout is enabled.
- Promotes canonical `transcripts/processed.json` and optional report.
- Records adapter invocation metadata, credential presence signal, and output provenance in stage metadata.
- polished transcript schema validation is mandatory.
- report output is config-gated.
## Skip and Resume Behavior
- Runner-level skip applies when already succeeded and not forced.
- Forced rerun can stale downstream succeeded stages via runner invalidation.
## Related Contracts And Tests
## Failure Behavior
- Fails on missing/invalid merged transcript, missing glossary, adapter error, invalid processed output shape (`segments` array required), or invalid report JSON when enabled.
## Tests to Inspect Before Changing
- `internal/stage/polish_test.go`
- `internal/adapters/audita/subprocess_test.go`
## Architectural Invariants
- Processed transcript must contain a top-level `segments` array.
- Report behavior is strictly config-gated.
- Stage output canonicalization always ends at `transcripts/processed.json`.
- [Audita](../integrations/audita.md) owns subprocess, validation, and failure
semantics.
- [Configuration](../config.md#pipeline) owns operator-selected Audita values.
- Implementation and tests: `internal/stage/polish.go`,
`internal/stage/polish_test.go`

View File

@@ -1,74 +1,60 @@
# Stage: prepare
## Purpose
Materialize all required session inputs into canonical local workspace paths and record input provenance in the session manifest.
## Inputs and Outputs
Inputs:
- `session.yml` (resolved session config)
- `pipeline.resolved.yml` (materialized from resolved pipeline config)
- `speakers.yml`
- `autocorrect.yml`
- `glossary.yml`
- audio source:
- local (`session.inputs.audio_dir` or `session.inputs.audio_files`), or
- S3 (`session.inputs.audio_s3.prefix`)
Materialize canonical current-session inputs before processing stages.
Outputs:
## Inputs
- resolved campaign, session, and pipeline configuration
- stable input files (`speakers`, `autocorrect`, `glossary`, `players`, `party`)
- one resolved local or S3 audio source
- enabled configured artifact input requirements for previous-session sources
## Outputs
- `inputs/campaign.yml`
- `inputs/session.yml`
- `inputs/pipeline.resolved.yml`
- `inputs/speakers.yml`
- `inputs/autocorrect.yml`
- `inputs/glossary.yml`
- `audio/*.flac` in session workdir
- `manifest.Inputs` records with checksums and source metadata
- `inputs/players.yml`
- `inputs/party.yml`
- `audio/*.flac`
- optional `previous/manifest.json`
- optional `previous/artifacts/**`
- deterministic `manifest.inputs` entries (checksums + provenance)
## Boundaries
Owns:
- Input path resolution and validation
- Local copy/materialization of configs and audio files
- S3 audio download to run-scoped spool, then copy into work audio dir
## Key Behavior
Does not own:
- Transcript generation/processing
- Archive publish behavior
- validates required config/store state.
- enforces local audio vs S3 audio mutual exclusivity.
- materializes S3 audio through spool/cache-aware logic.
- scans enabled configured artifact inputs for `narratio.previous_session.artifact.*` requirements.
- when previous requirements exist:
- clears managed `previous/` state;
- builds previous-cache remote plan;
- downloads previous manifest/artifacts;
- records previous inputs in `manifest.inputs`.
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `session.inputs.speakers_file`
- `session.inputs.autocorrect_file`
- `session.inputs.glossary_file`
- `session.inputs.audio_dir`
- `session.inputs.audio_files`
- `session.inputs.audio_s3.prefix`
- `pipeline.workspace.root`
- `pipeline.spool.root`
- `pipeline.storage.s3.bucket`
- `pipeline.storage.s3.root_prefix`
Required previous-session inputs fail when unavailable; optional missing inputs are skipped.
## External Adapters Used
- Object storage backend (`env.ObjectStore`) for S3 audio list/download when `audio_s3` is configured.
## Invariants
## State and Manifest Behavior
- Ensures workspace layout exists.
- Writes resolved config and input files to canonical `inputs/` paths.
- Records all prepared inputs into `manifest.Inputs` (sorted deterministically by kind/path).
- For S3 audio, records `S3Bucket`, `S3Key`, `S3Size`, `S3ETag`, and `SpoolPath` in each audio input record.
- only `prepare` hydrates canonical `previous/` cache state.
- managed previous artifacts are stored under `previous/artifacts/**` without
duplicate `artifacts/artifacts/` nesting.
- `manifest.inputs` ordering is deterministic (`kind`, `path`).
## Skip and Resume Behavior
- Runner-level skip applies when stage already `succeeded` and `--force` is not set.
- Stage itself is deterministic/idempotent for unchanged inputs (`copyFileIfChanged`, `writeBytesIfChanged`).
## Related Contracts And Tests
## Failure Behavior
- Fails on missing required files, invalid audio source combinations, no discoverable `.flac` files, duplicate audio basenames, missing object store for S3 mode, or S3 list/download failures.
## Tests to Inspect Before Changing
- `internal/stage/prepare_test.go`
- `internal/app/session_cli_test.go`
- `internal/config/load_validate_test.go`
## Architectural Invariants
- `audio_dir`/`audio_files` and `audio_s3` are mutually exclusive.
- Audio files must be `.flac`.
- Canonical `inputs/*` and `audio/*` paths are the durable source for downstream stages.
- [Configuration](../config.md) owns audio selection, stable input fields, and
previous-session settings.
- [Operations](../operations.md) owns physical input, audio, spool, cache, and
previous-state layout.
- [Storage Internals](storage.md) and [Artifact Internals](artifacts.md) explain
the internal collaborators.
- Implementation and tests: `internal/stage/prepare.go`,
`internal/stage/prepare_test.go`, `internal/audio/s3_audio_test.go`,
`internal/previouscache/*_test.go`

View File

@@ -0,0 +1,74 @@
# Stage: publish
## Purpose
Upload run/session outputs to object storage and atomically advance remote current state.
## Inputs
- successful preceding stages from the [canonical stage set](overview.md#pipeline-stage-set)
- invocation-scoped run files
- resolved publish output rules
- effective publish locks (static + remote merged lock set)
- durable previous-session cache files when present
## Outputs
- uploaded invocation record and selected publish outputs;
- uploaded durable previous-session cache files when present;
- updated remote current manifest; and
- remote current-run commit marker, written last.
Exact remote placement and the operator workflow belong in
[Operations](../operations.md#publish-workflow).
## Key Behavior
- stage can self-skip when publish disabled or run upload disabled.
- validates prerequisite stage success and object-store availability.
- collects a deterministic run file list plus run `manifest.json`, excluding
`audio/**` and the run-local `extract/notarius-output/**` staging bundle.
- keeps run-local Notarius receipt and stderr diagnostics eligible for the run
archive.
- resolves publish output sources through runtime artifact catalog and manifest-aware resolution.
- publishes extraction lanes only through explicit configured output rules;
neither run-local nor durable Notarius bundles are scanned or uploaded wholesale.
- selected artifact filter applies to configured artifact sources only.
- locked outputs are skipped intentionally (including required ones).
- optional missing outputs are skipped; required missing unlocked outputs fail.
- writes remote current manifest before current run pointer.
## Metadata Signals
Includes counts/lists for:
- run uploads
- published output uploads
- previous uploads
- skipped optional outputs
- skipped unselected outputs
- locked outputs
- current-state key paths
- `current_pointer_written`
## Invariants
- `current/run_id.txt` is the remote commit marker and is written last.
- run upload excludes `audio/**` and `extract/notarius-output/**`.
- `extract/notarius.receipt.json` and `extract/notarius.stderr.log` remain
eligible run-record diagnostics.
- publish locks are not overridden by `--force`.
The commit boundary and cleanup gate are normative architecture invariants; see
[Architecture](../policy/architecture.md#publish-commit-boundary).
## Related Contracts And Tests
- [Configuration](../config.md#publish-configuration-summary) owns output and
static-lock fields.
- [Operations](../operations.md#publish-locks) owns remote lock lifecycle and
physical remote state.
- [Artifact Internals](artifacts.md) explains source resolution and current-state
helpers.
- Implementation and tests: `internal/stage/publish.go`,
`internal/stage/publish_test.go`, `internal/app/operator_helpers_test.go`,
`internal/app/post_publish_cleanup_test.go`

View File

@@ -0,0 +1,42 @@
# Stage: render
## Purpose
Render Markdown transcript artifacts from normalized JSON transcripts via Seriatim.
## Inputs
- `narratio.transcript.final` (`transcripts/final.json`)
- `narratio.transcript.final_trimmed` (`transcripts/final.trimmed.json`)
## Outputs
- `narratio.transcript.final_markdown` -> `transcripts/final.md`
- `narratio.transcript.final_trimmed_markdown` -> `transcripts/final.trimmed.md`
## Key Behavior
- uses `pipeline.render` settings (enabled/format/title/booleans).
- resolves inputs manifest-first, then canonical fallback.
- writes run-local outputs first, then materializes canonical session outputs.
- records input provenance, output paths, adapter metadata, logs, and generated config refs.
- skips with stage metadata when `pipeline.render.enabled=false`.
## Failure Semantics
- missing normalized input fails with normalize rerun guidance.
- missing trimmed input fails with trim rerun guidance.
- adapter/subprocess failure fails stage.
- empty render output files fail validation.
## Invariants
- only `format: markdown` is supported.
- render stage owns production of built-in Markdown transcript sources.
## Related Contracts And Tests
- [Seriatim](../integrations/seriatim.md) owns render subprocess behavior.
- [Configuration](../config.md#pipeline) owns render fields and defaults.
- Implementation and tests: `internal/stage/render.go`,
`internal/stage/render_test.go`

View File

@@ -1,58 +1,36 @@
# Stage: transcribe
## Purpose
Generate per-speaker raw transcripts from prepared audio using WhisperX.
## Inputs and Outputs
Inputs:
- `audio/*.flac` prepared by `prepare`
Generate raw per-speaker transcripts from prepared audio using WhisperX.
Outputs:
- `transcripts/raw/<speaker>.json` for each input audio file
## Inputs
## Boundaries
Owns:
- Discovering prepared audio inputs
- Deriving speaker ids from audio basenames
- Parallel WhisperX invocation with bounded concurrency
- Validating produced JSON and promoting run-local outputs
- `audio/*.flac` from `prepare`
Does not own:
- Transcript merge/polish/normalize/trim/analyze
## Outputs
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.whisperx.transcribe_url`
- `pipeline.whisperx.language`
- `pipeline.whisperx.timeout`
- `pipeline.whisperx.retries`
- `pipeline.whisperx.retry_delay`
- `pipeline.whisperx.concurrency`
- `transcripts/raw/<speaker>.json`
## External Adapters Used
- WhisperX adapter (`env.WhisperX.Transcribe`).
## Key Behavior
## State and Manifest Behavior
- Uses run-local output paths under `runs/{run_id}/transcribe/outputs/...` when run layout is enabled.
- Validates each generated transcript JSON before promotion.
- Promotes canonical outputs to `transcripts/raw/*.json`.
- Records per-file metadata (attempts/status/duration/output path) in stage metadata.
- discovers prepared audio from manifest inputs or canonical audio directory.
- derives speaker ID from `.flac` basename.
- dispatches WhisperX requests through a bounded worker pool.
- validates each output as JSON.
- writes run-local outputs then materializes canonical transcript outputs.
## Skip and Resume Behavior
- Runner-level skip applies for previously succeeded stage unless forced.
- On forced upstream reruns, downstream succeeded stages can be marked `stale` by runner logic.
## Invariants
## Failure Behavior
- Fails if no prepared audio exists, duplicate speaker basenames are detected, adapter output path mismatches expected path, any output JSON is invalid, or one worker fails.
- Cancels in-flight workers after first terminal error.
- speaker basenames must be unique.
- output path returned by adapter must match requested output path.
- each successful output is validated before stage success.
## Tests to Inspect Before Changing
- `internal/stage/transcribe_test.go`
- `internal/app/whisperx_wiring_test.go`
## Related Contracts And Tests
## Architectural Invariants
- Speaker identity is derived from `.flac` basename and must be unique.
- Every successful speaker output must be valid JSON before promotion.
- Canonical raw transcript set is the only supported merge input surface.
- [WhisperX](../integrations/whisperx.md) owns HTTP, retry, timeout, and
cancellation semantics.
- [Configuration](../config.md#pipeline) owns concurrency and other
operator-selected values.
- Implementation and tests: `internal/stage/transcribe.go`,
`internal/stage/transcribe_test.go`

View File

@@ -1,75 +1,43 @@
# Stage: trim
## Purpose
Optionally trim the normalized transcript to session bounds; always produce a durable trimmed transcript.
## Inputs and Outputs
Inputs:
- `transcripts/normalized.json`
Produce a final-trimmed transcript. By default, the stage generates bounds and
applies a bounds-driven trim.
Outputs:
- `transcripts/trimmed.json` (or configured trim output path)
## Inputs
- `transcripts/final.json`
## Outputs
- `transcripts/final.trimmed.json` (or configured trim output path)
- when trim enabled: `artifacts/session_bounds.json`
## Boundaries
Owns:
- Trim-enabled switch behavior
- Bounds generation via Scriptorium artifact run
- Bounds validation against normalized transcript
- Keep-selector derivation and Seriatim trim invocation
- Copy-through behavior when disabled or bounds indicate unchanged transcript
## Key Behavior
Does not own:
- Upstream normalization
- Downstream artifact analysis
When `trim.enabled=true`:
- runs Scriptorium bounds artifact generation;
- optionally runs render-debug output generation;
- validates bounds payload against transcript;
- derives keep selector;
- either copies unchanged transcript or runs Seriatim trim;
- validates trimmed transcript and materializes bounds output.
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.trim.enabled`
- `pipeline.trim.output_path`
- `pipeline.trim.bounds.prompt_id`
- `pipeline.trim.bounds.profile_id`
- `pipeline.trim.bounds.timeout`
- `pipeline.trim.bounds.output_path`
- `pipeline.trim.bounds.transcript_input_name`
- `pipeline.trim.bounds.render_debug`
- `pipeline.trim.bounds.render_output_path`
- `pipeline.seriatim.binary`
- `pipeline.seriatim.timeout`
- `pipeline.scriptorium.binary`
- `pipeline.scriptorium.config_path`
- `pipeline.scriptorium.timeout`
When `trim.enabled=false`:
- copies normalized transcript to trimmed output.
## External Adapters Used
- Scriptorium adapter:
- optional `RenderArtifact` for bounds debug render
- `RunArtifact` for bounds output
- Seriatim adapter:
- `Trim` when bounds indicate trimming is required
## Invariants
## State and Manifest Behavior
- Reads normalized transcript from normalize manifest outputs when available; falls back to canonical path.
- Uses run-local outputs/logs/reports/config/scratch paths when run layout is enabled.
- Promotes canonical trimmed transcript; promotes session bounds when trim enabled.
- Records bounds diagnostics, trim action, keep selector, and adapter metadata.
- normalized transcript is required input.
- bounds output exists only in enabled trim path.
- render-debug output is diagnostic and not a declared stage output.
## Skip and Resume Behavior
- Runner-level skip applies when already succeeded and not forced.
- Forced reruns can stale downstream succeeded stages.
- When `trim.enabled=false`, stage still succeeds by copying normalized to trimmed output.
## Related Contracts And Tests
## Failure Behavior
- Fails on missing/invalid normalized transcript.
- With trim enabled, fails on missing adapters/config, bounds generation/validation errors, invalid bounds JSON, invalid range/segment ids, trim adapter failures, or invalid trimmed output.
## Tests to Inspect Before Changing
- `internal/stage/trim_test.go`
- `internal/adapters/scriptorium/subprocess_test.go`
- `internal/adapters/seriatim/subprocess_test.go`
## Architectural Invariants
- Trim never falls back to processed transcript; normalized transcript is required input.
- `session_bounds` output exists only for enabled trim path.
- Render-debug artifacts are diagnostics and not declared stage outputs.
- [Scriptorium](../integrations/scriptorium.md) owns bounds generation and
debug-render subprocess behavior.
- [Seriatim](../integrations/seriatim.md) owns transcript trimming behavior.
- [Configuration](../config.md#pipeline) owns trim fields and defaults.
- Implementation and tests: `internal/stage/trim.go`,
`internal/stage/trim_test.go`

View File

@@ -1,71 +1,48 @@
# Internal: Storage
## Purpose
Document Narratio's remote storage backend contracts and implementations under `internal/adapters/storage`.
## Inputs and outputs
Inputs:
- Resolved storage config (`pipeline.storage.*`).
- Bucket-relative object keys and local file paths from stage/app orchestration.
Explain the object-store interface and S3 implementation used by Narratio.
Remote key layout and lifecycle belong in [Operations](../operations.md), while
operator-selected storage fields and credential mechanisms belong in
[Configuration](../config.md).
Outputs:
- Listed/downloaded/uploaded object metadata (`ObjectInfo`).
- Existence checks and storage-layer errors.
## Primary Contract
## Boundaries
Owns:
- Remote object-store interface and implementation details.
- S3 client wiring and API calls.
- Object key normalization and upload/download/list primitives.
`storage.ObjectStore` interface:
Does not own:
- Session/run prefix semantics.
- Archive commit order semantics.
- Manifest updates.
- `List(ctx, prefix)`
- `Download(ctx, key, localPath)`
- `Upload(ctx, localPath, key, opts)`
- `Exists(ctx, key)`
## Config fields used
- `pipeline.storage.backend`
- `pipeline.storage.s3.bucket`
- `pipeline.storage.s3.region`
- `pipeline.storage.s3.endpoint`
- `pipeline.storage.s3.force_path_style`
- `pipeline.storage.s3.access_key_id_env`
- `pipeline.storage.s3.secret_access_key_env`
Key invariant:
- callers pass full bucket-relative keys;
- storage implementations do not infer campaign/session/run prefixes.
## External adapters used
Storage package contracts:
- `ObjectStore` (active remote object-store boundary): `List`, `Download`, `Upload`, `Exists`.
- `Backend` (archive request boundary): currently implemented with `NoopBackend` only.
## Composition
Implementations:
- `S3Backend`: AWS SDK-backed `ObjectStore` implementation.
- `FakeBackend`: deterministic test `ObjectStore` and archive backend.
- `NoopBackend`: deterministic no-op archive backend for compatibility wiring.
`NewObjectStoreFromConfig` constructs the S3-backed implementation from
resolved configuration. The application loads configured filesystem secrets
before calling it. The storage adapter consumes already-resolved values; it does
not own discovery, defaults, or configuration validation.
## State and manifest behavior
- Storage implementations are stateless with respect to manifest/session lifecycle.
- Caller supplies fully-qualified bucket-relative keys.
- Storage layer does not infer campaign/session/run/root-prefix semantics.
- Caller controls publish ordering; storage layer executes individual operations in the order invoked.
## S3 Backend Behavior
## Skip and resume behavior
- No storage-level skip/resume behavior.
- Skip/resume decisions are made by stage/app logic before storage calls occur.
- normalizes object keys.
- `List` paginates and returns normalized `ObjectInfo`.
- `Download` writes local files with parent directory creation.
- `Upload` streams local file and returns remote metadata.
- `Exists` maps not-found responses to `false`.
## Failure behavior
- `NewObjectStoreFromConfig` fails when no remote backend is configured or required S3 config is missing.
- `S3Backend` constructor fails when required bucket is missing or AWS client setup fails.
- CRUD operations return contextual errors (including not-found behavior via `Exists`).
- Key normalization is applied before operations (`\\` to `/`, leading slash trimmed).
## Invariants
## Tests to inspect before changing
- `internal/adapters/storage/factory_test.go`
- `internal/adapters/storage/s3_backend_test.go`
- `internal/adapters/storage/fake_test.go`
- `internal/adapters/storage/keys_test.go`
- `internal/adapters/storage/archive.go` + consumers in stage tests (`prepare`, `archive`)
- storage layer is stateless regarding manifest/stage progression.
- publish ordering semantics are owned by stage/app code, not storage adapters.
## Architectural invariants
- Callers pass full bucket-relative keys.
- Storage backends must not prepend or infer narratio prefixes.
- Remote transport details remain isolated to storage adapter implementations.
## Implementation And Tests
- Contract and S3 adapter: `internal/adapters/storage`
- Composition: `internal/app/object_store.go`
- Tests: `internal/adapters/storage/*_test.go`,
`internal/app/object_store_test.go`

View File

@@ -1,68 +1,74 @@
# Workspace internals
# Internal: Workspace
## Purpose
Define the local durable and run-local workspace model used by stages, manifests, resume, and archive.
## Inputs and Outputs
Inputs:
- `pipeline.workspace.root`
- `session.campaign`
- `session.session_id`
- generated `run_id`
Explain the helpers that construct local session and run paths, coordinate
single-writer access, and confine cleanup. The authoritative physical layout and
retention workflow belong in [Operations](../operations.md#local-state-layout).
Outputs:
- Session manifest at `{workspace.root}/work/{campaign}/{session_id}/manifest.json`
- Run manifest at `{workspace.root}/work/{campaign}/{session_id}/runs/{run_id}/manifest.json`
- Canonical durable session directories and run-local stage trees
## Path Ownership
## Boundaries
Owns:
- Session-level path layout (`inputs/`, `audio/`, `transcripts/`, `artifacts/`, `reports/`, `logs/`, `config/`, `current/`, `runs/`)
- Run-local stage sandbox layout under `runs/{run_id}/{stage}/`
- Session lock acquisition/release (`.lock`)
`internal/artifacts` owns canonical session, run, spool, cache, and
previous-cache path construction. `SessionPathsFor` provides the session-scoped
path model, and layout creation goes through `EnsureLayoutFor`. Callers should
consume those helpers instead of rebuilding relative paths.
Does not own:
- Stage business logic
- Remote archive semantics (documented in `stage-archive.md`)
- CLI argument parsing
`internal/pathsafe` and application cleanup helpers enforce confinement for
relative destinations and deletion targets.
## Config Fields Used
- `pipeline.workspace.root`
- `pipeline.workspace.cleanup_after_archive`
- `pipeline.spool.root`
- `pipeline.spool.delete_audio_after_archive`
- `session.campaign`
- `session.session_id`
## Run-Local Stage Layout
## External Adapters Used
None directly in this subsystem. Stages may use object storage adapters and then write local outputs into this layout.
`internal/stage/run_local.go` maps stage outputs and diagnostics into an
invocation-scoped layout. Successful outputs are validated and atomically
materialized into canonical session paths before stage success. Managed
previous-session cache paths remain session-durable and are never redirected
into run-local output space.
## State and Manifest Behavior
- Session state is persisted in the session manifest (`manifest.Manifest`).
- Invocation history is persisted per run in run manifests under `runs/{run_id}/manifest.json`.
- During each run, stage outputs are often written run-local first (`runs/{run_id}/{stage}/outputs/...`) and promoted to canonical session paths after stage success.
- `manifest.Artifacts` entries record `ProducerRunID` for durable outputs.
- For S3 audio sessions, `prepare` records spool/work paths and S3 provenance in `manifest.Inputs`.
Extraction uses run-local receipt, stderr, and output-root helpers, then
promotes the validated external bundle to the unique immutable Notarius bundle
path supplied by `internal/artifacts`. `internal/fileops.PromoteDirectory`
copies only regular files and directories to a same-filesystem temporary
sibling. Source traversal uses confined directory handles and identity checks
so replacing an inspected root, directory, or file is rejected rather than
followed. The completed tree is atomically renamed without replacing an
existing destination. Exact physical paths belong in
[Operations](../operations.md#extraction-workflow).
## Skip and Resume Behavior
- Skip/resume decisions are made in `internal/app` (`run_control.go`, `resume.go`) using stage status in the session manifest.
- `--force` reruns selected stages and marks downstream previously-succeeded stages as `stale`.
- Workspace layout is idempotent (`EnsureLayoutFor`) and reused across runs.
## Locking
## Failure Behavior
- Failures preserve manifests and run-local files for inspection.
- Lock conflicts fail fast via `ErrLockConflict`.
- Cleanup can fail post-archive; failure is recorded in archive stage metadata and returned by the run.
`artifacts.LocalStore` enforces the single-writer session lock via `.lock`
(`ErrLockConflict` on contention).
## Tests to Inspect Before Changing
- `internal/artifacts/local_test.go`
- `internal/stage/run_local_test.go`
- `internal/app/run_control_test.go`
- `internal/app/resume_run_stage_test.go`
- `internal/app/post_archive_cleanup_test.go`
## Cleanup Semantics
## Architectural Invariants
- Session root is campaign-aware: `{workspace.root}/work/{campaign}/{session_id}`.
- Run roots are always nested: `runs/{run_id}` under the session root.
- Run-local output promotion must end in canonical session paths.
- Cleanup only targets run-scoped directories and must never delete configured root directories.
Automatic post-publish cleanup:
- only runs when publish actually executed and succeeded;
- requires `uploaded=true` and `current_pointer_written=true` metadata;
- consumes the resolved cleanup policy described in
[Configuration](../config.md);
- refuses unsafe deletes (root delete, out-of-root delete, symlink paths).
Manual cleanup uses the same scoped-target checks. Invocation syntax and exact
deletion scope belong in [CLI](../cli.md#clean) and
[Operations](../operations.md#cleanup).
## Invariants
- campaign-aware session root is mandatory.
- manifest-driven stage state is durable across runs.
- cleanup guardrails prevent destructive root/out-of-scope deletion.
## Implementation And Tests
- Path model and local store: `internal/artifacts/paths.go`,
`internal/artifacts/local.go`
- Run-local materialization: `internal/stage/run_local.go`
- Immutable bundle promotion: `internal/fileops/directory.go`
- Cleanup confinement: `internal/app/cleanup_targets.go`,
`internal/app/post_publish_cleanup.go`
- Tests: `internal/artifacts/paths_model_test.go`,
`internal/artifacts/local_test.go`, `internal/stage/run_local_test.go`,
`internal/fileops/directory_test.go`,
`internal/app/cleanup_targets_test.go`,
`internal/app/post_publish_cleanup_test.go`

View File

@@ -1,213 +1,322 @@
# Operations
# Operations Guide
This guide describes the implemented operator lifecycle for Narratio.
Operator workflow for running, recovering, and publishing Narratio sessions.
For field-level configuration, see [docs/config.md](./config.md). For full command/flag reference, see [docs/cli.md](./cli.md).
For command syntax, see [docs/cli.md](./cli.md). For field-level config, see [docs/config.md](./config.md).
## Normal workflow (S3-first path)
## Campaign and Session Selection
1. Upload session `.flac` files to object storage under the configured session audio prefix.
2. Run Narratio:
Campaign selection priority:
- `--campaign-file`
- `--campaign`
- `pipeline.campaigns.default_campaign_id`
Session source priority:
- `--session`
- local default search paths
- remote session object (S3) when local session file is not found and storage is configured
## Session Initialization
Use `session init` to generate a concrete session file for local or remote use.
Local file:
```bash
narratio run --session-id 2026-04-04
narratio session init 2026-04-04 --output ./session.yml --date 2026-04-04 --title "Session 12"
```
3. Read success output:
- `narratio run: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
- use `manifest=<path>` with `status` for inspection.
Notes:
- default config/session discovery applies unless `--config` and `--session` are passed.
- S3 audio mode requires `session.inputs.audio_s3.prefix` and valid object-store access.
## Restore workflow
Use restore when local durable session state is missing or stale and archive current state is authoritative.
Dry-run (no local writes):
Remote session object:
```bash
narratio restore --session-id 2026-04-04 --dry-run
narratio session init 2026-04-04 --remote --force
```
Execution:
If `campaign.yml` sets `session_template_file`, `session init` renders it. Template variables must resolve to concrete values.
Campaigns must provide stable input files for speakers, autocorrect, glossary, players, and party. Session files may override those paths for one session. The `prepare` stage materializes them under `inputs/`; configured Scriptorium artifacts can reference prepared `players`, `party`, and `glossary` files with `narratio.input.players`, `narratio.input.party`, and `narratio.input.glossary`.
## Standard Session Workflow
1. Select pipeline/campaign/session config.
2. Validate session readiness:
```bash
narratio restore --session-id 2026-04-04
narratio session validate 2026-04-04
```
Post-restore analyze rerun pattern:
3. (Optional) inspect stage decisions:
```bash
narratio run-stage --session-id 2026-04-04 --force analyze
narratio session plan 2026-04-04
```
Restore source-of-truth:
- remote commit marker: `current/run_id.txt`
- remote current manifest: `current/manifest.json`
4. Run the pipeline:
Restore default scope:
- includes `manifest.json`, `transcripts/**`, `artifacts/**`
- includes `audio/**` only with `--include-audio`
- excludes `runs/**`, `logs/**`, `reports/**`, `config/**`, `inputs/**`, and `current/**` (except remote `current/manifest.json` as source)
```bash
narratio run 2026-04-04
```
## Local filesystem layout and state artifacts
5. Check state:
```bash
narratio session status 2026-04-04
```
## Stage Execution and Continuation Behavior
Canonical stage order:
1. `prepare`
2. `transcribe`
3. `merge`
4. `polish`
5. `normalize`
6. `trim`
7. `extract`
8. `render`
9. `analyze`
10. `publish`
11. `notify`
Execution rules:
- succeeded stages are skipped unless `--force` is set;
- `run` continues interrupted or partially completed sessions by running non-succeeded stages;
- forcing an upstream stage marks succeeded downstream stages as `stale` before
the replacement runs; and
- an executed failure, changed self-skip, or success that replaces a different
effective upstream outcome also marks succeeded downstream stages stale. A
repeated self-skip with the same reason and no outputs is stable and does not
perpetually rerun downstream work.
Single-stage execution:
```bash
narratio run-stage normalize 2026-04-04 --force
```
## Artifact Selection
`--artifacts` can be used on `run`, `run-stage`, `analyze`, and `publish`.
Selection behavior:
- validates names against `pipeline.scriptorium.artifacts`;
- filters analyze execution to selected configured artifacts;
- filters publish rules for `narratio.artifact.<name>` sources only;
- does not suppress built-in transcript, bounds, or explicitly configured
`narratio.extraction.<name>` publish sources; and
- never partially selects Notarius lanes.
## Extraction Workflow
When Notarius is omitted or disabled, `extract` records an explicit skipped
outcome with reason `notarius_disabled` and no outputs. A later invocation
reconsiders the skipped stage, so enabling Notarius does not require force.
When Notarius extraction is enabled, the stage consumes the final trimmed JSON
and preserves the complete validated Notarius bundle at:
- `artifacts/notarius/{narratio_run_id}/`
The directory is immutable once promoted. Configured lanes become
`narratio.extraction.<name>` sources for Scriptorium and explicit publish rules;
the bundle and `index.json` are retained for audit and resume validation but
are not selectable or published implicitly.
Starting a replacement clears the previous extraction payload from the current
session-stage record. If that replacement fails or self-skips, the current
record does not fall back to the earlier outputs. The earlier run manifest and
immutable bundle remain available for inspection, but downstream resolution
requires a new current successful extraction record.
Atomic Notarius bundle promotion is supported on Linux, macOS, and Windows.
On other operating systems, extraction fails before copying the bundle into a
temporary promotion tree because Narratio has no verified atomic no-replace
directory primitive there. This is an extraction limitation, not a broader
platform-support guarantee for every Narratio workflow.
Run-local diagnostics are:
- `runs/{run_id}/extract/notarius.receipt.json`
- `runs/{run_id}/extract/notarius.stderr.log`
- `runs/{run_id}/extract/notarius-output/` before durable promotion
The run-record upload excludes the complete
`extract/notarius-output/**` subtree. The receipt and stderr files remain
eligible run-record diagnostics. The durable bundle is never scanned for
implicit publication; only lanes named by explicit `pipeline.publish.outputs`
rules are uploaded.
To intentionally replace the current extraction result, run:
```bash
narratio run-stage extract 2026-04-04 --force
```
Narratio automatically reruns extraction when its recorded invocation contract
or durable output validation changes. It cannot fingerprint configuration
files, profiles, prompts, modules, or references loaded transitively by
Notarius. Force extraction after changing any of those inputs, even when the
top-level Narratio and Notarius config paths remain the same. A forced extract
marks successful downstream stages stale. Ordinary extraction failures or
outcome changes also stale affected downstream stages, while an identical
repeated `notarius_disabled` self-skip does not repeatedly invalidate them.
## Publish Workflow
Run publish only:
```bash
narratio publish 2026-04-04
```
Equivalent:
```bash
narratio run-stage publish 2026-04-04 --force
```
Publish commit model:
- uploads eligible run files under `{session_prefix}/runs/{run_id}/`, excluding
audio and the run-local Notarius staging bundle;
- uploads configured published outputs, including only explicitly configured
extraction lanes;
- uploads `previous/**` cache files when present;
- writes `current/manifest.json`;
- writes `current/run_id.txt` last.
`current/run_id.txt` is the remote current-state commit marker.
## Publish Locks
Lock sources:
- static locks in `pipeline.publish.locks`
- mutable remote locks in `{session_prefix}/locks.yml`
Effective lock rules:
- static and remote locks are merged;
- static locks win on source collisions;
- locked outputs are intentional skips;
- lock add/remove commands mutate only remote lock state.
Examples:
```bash
narratio session locks 2026-04-04
narratio session locks add 2026-04-04 narratio.artifact.session_recap --reason "manual edits" --force
narratio session locks remove 2026-04-04 narratio.artifact.session_recap
```
## Restore Workflow
Use restore when local durable session state is missing or stale and remote committed current state is authoritative.
Dry run:
```bash
narratio session restore 2026-04-04 --dry-run
```
Apply:
```bash
narratio session restore 2026-04-04
```
Default restore scope:
- `manifest.json`
- `transcripts/**`
- `artifacts/**`
- `previous/**` when needed by configured previous-session artifact inputs
Optional:
- `--include-audio` to include `audio/**`
- `--force` to overwrite local conflicts
Restore writes an execution report at `reports/restore-latest.json`.
## Local State Layout
Session root:
- `{workspace.root}/work/{campaign}/{session_id}/`
Primary state:
- `manifest.json`: session-level stage state.
- `runs/{run_id}/manifest.json`: invocation-level state.
- `.lock`: session lock while a modifying command is active.
- `{workspace.root}/work/{campaign}/{session_id}`
Canonical session directories:
- `inputs/`
- `audio/`
- `transcripts/`
- `artifacts/`
- `reports/`
- `logs/`
- `config/`
- `current/`
- `runs/`
Durable session paths:
Run-local stage directories:
- `runs/{run_id}/{stage}/` with stage-local `outputs/`, `logs/`, `reports/`, `config/`, `scratch/`.
- `manifest.json`
- `inputs/**`
- `audio/**`
- `transcripts/**`
- `artifacts/**`
- `previous/**`
- `reports/**`
- `logs/**`
- `config/**`
- `runs/**`
Behavior:
- directory creation is idempotent.
- stage outputs are generally generated run-local first, then promoted to canonical paths on success.
- restore installs downloaded files to canonical session paths and does not recreate historical run sandboxes.
Validated Notarius bundles live below `artifacts/notarius/{run_id}/`; receipt,
stderr, and pre-promotion output remain in the producing run's `extract`
directory as described in [Extraction Workflow](#extraction-workflow).
## Analyze artifact execution lifecycle
Run-local layout:
Analyze executes configured artifacts from `pipeline.scriptorium.artifacts`.
- `runs/{run_id}/{stage}/outputs`
- `runs/{run_id}/{stage}/logs`
- `runs/{run_id}/{stage}/reports`
- `runs/{run_id}/{stage}/config`
- `runs/{run_id}/{stage}/scratch`
Execution model:
- executable set = enabled artifacts, filtered by `--artifacts` when provided.
- artifact-to-artifact dependencies are declared via `depends_on`.
- selected artifacts run in deterministic dependency order.
- after each successful artifact run, output is promoted to configured canonical `output_path`.
Spool layout (runtime/transient):
Configured artifact source reuse:
- a non-executable configured artifact can satisfy inputs if its configured output file already exists and is valid.
- reused configured artifact provenance is `filesystem.disabled_artifact_output`.
- `{spool.root}/{campaign}/{session_id}/{run_id}/...`
- restore audio spool under `{spool.root}/{campaign}/{session_id}/restore/audio`
`--artifacts` behavior:
- accepted on `run`, `resume`, and `run-stage analyze`.
- filters analyze execution only; does not force stage rerun.
Cache layout (durable S3 audio cache):
## Remote archive layout and publish contract
- `{cache.root}/s3/{bucket}/...`
When archive is enabled and run upload is enabled, archive publishes under:
## Cleanup
- session prefix: `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/`
- run prefix: `{session_prefix}/runs/{run_id}/`
Archive uploads:
- run record files from run root (excluding `audio/`).
- promoted files from explicit `archive.promote_artifacts` rules.
Publish order:
1. upload `current/manifest.json`
2. upload `current/run_id.txt` last
`current/run_id.txt` is the remote commit marker.
Archive promotion is explicit and source-based:
- Narratio does not auto-promote all generated analyze artifacts.
- each rule resolves `source` through the artifact resolver/catalog model, then uploads to `dest`.
- missing required promotion sources fail archive stage.
- missing optional promotion sources are skipped.
- invalid resolved artifacts fail archive stage.
## Resume, retry, restore, and safe rerun behavior
Default skip:
- `run` and `run-stage` skip already-succeeded stages unless `--force` is set.
Resume:
- `resume` starts at first non-succeeded stage.
- `resume --force` runs full stage order.
Restore conflict policy:
- restore classifies local differences as conflicts.
- without `--force`, restore fails when conflicts exist.
- with `--force`, conflicting local files are overwritten by remote archive files.
Forced reruns:
- force-rerunning an upstream succeeded stage marks downstream succeeded stages as `stale`.
Safe rerun pattern:
1. rerun the changed stage with `--force`.
2. run `resume` to rebuild downstream stages.
## Cleanup behavior
Cleanup is considered only when archive stage executed and succeeded.
Cleanup toggles:
- `pipeline.spool.delete_audio_after_archive=true` deletes run-scoped spool audio.
- `pipeline.workspace.cleanup_after_archive=true` deletes run-scoped local run directory.
Cleanup eligibility gates:
- archive enabled
- archive run upload enabled
- run record upload completed
- current pointer write completed (`current/run_id.txt` written)
No cleanup for failed/incomplete/unarchived/archive-skipped runs.
## Failure and recovery playbooks
After run failure, Narratio keeps:
- session manifest
- run manifest
- run-local artifacts/logs/config/reports
Failed or incomplete runs remain local-only.
After restore failure:
- already-installed restore files remain in place.
- restore does not roll back prior successful installs.
- existing local manifest is preserved if restored manifest validation/install fails.
Recommended recovery:
1. inspect state:
Session-scoped cleanup:
```bash
narratio status --manifest <manifest-path>
narratio clean 2026-04-04
```
2. for restore-specific checks, run:
Global cleanup:
```bash
narratio restore --session-id 2026-04-04 --dry-run
narratio clean --all
```
3. fix root cause (config/input/credentials/storage/service availability).
4. continue with `resume`, or targeted `run-stage --force` followed by `resume`.
Dry-run and cache variants:
## Restore report
```bash
narratio clean 2026-04-04 --dry-run --clear-cache
narratio clean --all --dry-run --clear-cache
```
Non-dry-run restore writes a durable report at:
- `reports/restore-latest.json`
Rules:
Report content includes:
- identity (`campaign`, `session_id`, `run_id`)
- mode flags (`dry_run`, `force`, `include_audio`)
- plan counts and execution counts
- per-action status
- `clean` deletes work/spool session state;
- cache is preserved unless `--clear-cache` is set;
- automatic post-publish cleanup is gated by successful publish commit plus:
- `pipeline.spool.delete_audio_after_publish=true`
- `pipeline.workspace.cleanup_after_publish=true`
Dry-run does not write restore report files.
## Operational Caveats
## Operational caveats
- `status` requires explicit `--manifest`; there is no session-id lookup command.
- local and S3 audio input modes are mutually exclusive.
- archive publish requires upstream stages through `analyze` to be `succeeded`.
- required promotion rules can fail when selected analyze artifacts did not generate a required file path.
- restore requires configured remote object storage and committed remote current state.
- Local and S3 audio modes are mutually exclusive.
- Publish requires prerequisite stages through `render` and `analyze` to be succeeded.
- Markdown publish defaults require render outputs (`transcripts/final.md` and `transcripts/final.trimmed.md`).
- Restore requires configured object storage and committed remote current state.
- Storage-backed commands load filesystem secrets before object-store initialization.

241
docs/policy/architecture.md Normal file
View File

@@ -0,0 +1,241 @@
# Architecture
This document defines Narratio's intended high-level architecture and the
invariants that changes must preserve. Implemented component details belong in
the [Internal Overview](../internal/overview.md) and its linked documents.
Significant architectural decision history belongs under `docs/adr/` when such
records exist.
## System Shape
Narratio is a small Go application that turns D&D session audio into polished
transcripts and generated session artifacts. It is an explicit, stage-driven
orchestrator, not a general workflow engine.
Narratio coordinates specialized external systems rather than reimplementing
their domains:
- WhisperX performs transcription;
- Seriatim performs deterministic transcript processing and rendering;
- Audita performs transcript correction and polishing;
- Notarius extracts validated structured artifact bundles; and
- Scriptorium executes prompts and produces configured artifacts.
Narratio owns orchestration, configuration resolution, session and run state,
artifact and path modeling, manifest persistence, stage sequencing, resume,
restore, cleanup gates, and publish semantics. External contracts are defined
in the [integration documentation](../integrations/).
The pipeline has one canonical ordered stage set. Configuration may enable,
disable, or parameterize supported behavior, but it must not turn that sequence
into an arbitrary DAG or hide orchestration in generic workflow abstractions.
The implemented stage inventory belongs in the
[Internal Overview](../internal/overview.md).
Narratio is contract-first without being abstraction-heavy. Interfaces and
extension points should protect demonstrated boundaries. New abstraction is not
itself an architectural goal.
## Ownership And Dependency Direction
The application boundary owns command dispatch, configuration selection,
production composition, session locking, and top-level lifecycle. It may depend
on concrete implementations to assemble a run.
Stage orchestration expresses intent in Narratio-level data and interfaces.
Stages may depend on configuration, manifest, artifact, path, and adapter
contracts, but they must not depend on transport-specific request types,
subprocess argument construction, cloud SDK types, or downstream tool internals.
Adapters translate between Narratio contracts and external systems. They own
HTTP, subprocess, notification, and object-storage mechanics, including command
construction, transport behavior, provider response handling, and external
error adaptation. External dependency types must remain inside the adapter that
owns them unless that dependency is the adapter's explicit public contract.
WhisperX HTTP behavior, Seriatim, Audita, Notarius, and Scriptorium command
construction, notification transport, and object-storage SDK details remain
behind these boundaries.
State and path services must not infer stage policy. Storage implementations
receive explicit bucket-relative keys and do not infer campaign, session, run,
or root-prefix semantics. Manifest persistence records transitions but does not
choose orchestration policy. Artifact resolution identifies and validates
artifacts but does not execute producers.
Dependencies should remain narrow and point toward Narratio-owned contracts.
Prefer the Go standard library. Add an external dependency only when it provides
a clear correctness, security, interoperability, or complexity benefit, and
confine it to the boundary that needs it.
## Stage Boundaries
Each stage has one explicit responsibility and declares:
- required input state;
- produced output state;
- configuration it consumes;
- external adapters it uses;
- manifest references and metadata it reads or writes;
- skip, force, invalidation, and resume behavior; and
- failure behavior.
Stages write and validate run-local results before materializing canonical
outputs where that distinction applies. A stage is complete only after its
required outputs have been written, validated, and recorded in durable manifest
state. Later stages depend on recorded success and artifact resolution, not
merely on incidental files existing on disk.
A failed or interrupted stage must not be presented as successful. Failure
should preserve enough local state and diagnostics for inspection, recovery,
and resume. Forcing an upstream stage invalidates succeeded downstream work
according to the canonical stage order.
A stage may explicitly self-skip with a stable reason and no outputs. That
outcome is persisted, clears older outputs owned by the stage, and is
reconsidered on a later invocation. A stage may also validate whether an
otherwise successful recorded result is still resumable; an obsolete result
is staled and rerun, while an unsafe condition that prevents a sound decision
stops execution.
Shared behavior should live behind a narrow service or helper with one clear
owner. Stages must not reach across boundaries or reproduce adapter, manifest,
artifact, or path policy ad hoc.
## Manifest, Resume, And Restore
The session manifest is the durable ledger for progress across invocations. It
records session and run identity, stage state, input and output references,
diagnostic references, checksums or provenance where useful, and non-secret
adapter and publish metadata.
Resume and skip decisions are manifest-driven. Filesystem state may be
inspected and validated, but file presence alone does not replace recorded
stage state. Invocation-scoped run records provide an audit of one execution;
they do not replace the session manifest as progress authority.
Restore treats committed remote current state as its authority. It must plan
deterministically, confine remote-to-local paths, protect local conflicts, and
install the validated session manifest after other restored durable files. The
physical workflow and recovery procedures belong in
[Operations](../operations.md).
## Configuration
Configuration is strict, explicit, centralized, and operator-oriented.
- YAML decoding rejects unknown fields.
- Defaults are centralized and testable.
- Empty configured values do not silently replace meaningful defaults.
- Validation rejects invalid composition before stage execution where
practical.
- Session templating remains narrow and deterministic rather than becoming a
general configuration language.
- Secret values are supplied indirectly and are not persisted in ordinary
configuration.
Narratio must not become a second configuration system for downstream tools.
External systems own their runtime defaults wherever practical; Narratio passes
the paths required by its stage contracts and explicit operator overrides. The
field-level contract and credential-supply mechanisms belong in
[Configuration](../config.md).
## Artifacts, Paths, And Storage
Artifact identities and local and remote paths are application contracts.
Canonical helpers own workspace, spool, cache, session, run, input, transcript,
artifact, log, report, configuration, and publish-current paths. Callers must
not reconstruct canonical paths through scattered string concatenation.
Artifact resolution is deterministic and manifest-aware. Producers materialize
canonical outputs before reporting success, and consumers resolve declared
artifact identities rather than infer files from unrelated directory contents.
External artifact bundles become current only through validated immutable
promotion and manifest records; directory presence alone never establishes
availability.
Writes, moves, replacements, and deletions must use narrow, explicit,
root-confined destinations. Symlinks, traversal, broad roots, and ambiguous
relative destinations must not expand the scope of an operation. Cleanup is
permitted only through explicit operator action or configured post-publish
gates, and it must preserve durable cache unless cache removal is explicitly
requested.
Physical layout, retention, and operational lifecycle belong in
[Operations](../operations.md). Logical external formats and durable integration
contracts belong under [Integrations](../integrations/).
## Publish Commit Boundary
Publish has one explicit remote commit boundary. A remote run becomes current
only after Narratio has successfully uploaded the run record, required published
outputs, `current/manifest.json`, and finally `current/run_id.txt`.
`current/run_id.txt` is the commit marker and must be written last. Failed,
incomplete, skipped, or uncommitted publish attempts must not be presented as
current remote state. Publish locks remain authoritative and are not bypassed by
a forced run.
Automatic local cleanup is permitted only after a successful publish commit,
only when explicitly configured, and only through the path-safety guardrails.
## Security, Privacy, And Diagnostics
Narratio handles private campaign material. Transcripts, prompts, generated
artifacts, reports, logs, manifests, and diagnostic files are potentially
sensitive.
Raw secrets must not be stored in pipeline, campaign, or session YAML or written
to manifests, logs, generated configuration, reports, publish metadata,
documentation, or examples. Secrets enter through configured environment
variable names or secret-file references. Diagnostics should avoid transcript
and prompt content unless a deliberate, bounded inspection mechanism requires
it.
Logs, reports, generated invocation files, generated configuration, and render
debug files are diagnostics, not canonical pipeline products. They should be
durable and discoverable where configured, and manifest references must preserve
the distinction between diagnostics and artifacts.
Documentation security rules belong in the
[Documentation Policy](documentation.md). Credential supply belongs in
[Configuration](../config.md), while permissions, sensitive runtime-artifact
handling, and recovery belong in [Operations](../operations.md).
## Determinism And Testability
Narratio prefers deterministic behavior where practical, including stable local
and remote layouts, sorted operation order, predictable generated
configuration, repeatable command construction, deterministic artifact
resolution, and reproducible planning.
Run IDs and timestamps may be intentionally variable, but surrounding behavior
must remain controllable in tests. Core behavior should be testable without live
external services; expensive, nondeterministic, destructive, or external
boundaries should be replaceable with focused test doubles. General testing
philosophy and sufficiency rules belong in the [Testing Policy](testing.md).
## Documentation And Decision Records
Documentation follows the [Documentation Policy](documentation.md). Current
behavior belongs in its canonical user, operator, integration, architecture, or
internal owner. Proposed behavior and implementation status belong under
`docs/roadmap/`.
Significant architectural decisions may be recorded under `docs/adr/` using the
format and lifecycle defined by the documentation policy. ADR acceptance does
not establish that a decision has been implemented.
## Architectural Non-Goals
Narratio does not aim to provide:
- a generic DAG or workflow engine;
- a replacement configuration layer for WhisperX, Seriatim, Audita,
Scriptorium, or other downstream tools;
- a storage abstraction broader than the needs of this pipeline;
- stage logic coupled directly to cloud SDKs, transports, subprocess details,
or downstream implementation internals;
- raw-secret persistence;
- implicit cross-stage behavior that bypasses manifest and artifact contracts;
or
- a prompt-authoring system.

View File

@@ -0,0 +1,148 @@
# Documentation Policy
## Purpose
This policy assigns each documentation topic to one canonical owner. Its goal is
to keep Narratio documentation accurate, concise, discoverable, and resistant
to drift for users, operators, developers, integrators, and LLM coding agents.
## Core Rules
### One Canonical Owner
Each authoritative fact belongs in one document. A non-owning document may give
a short, stable summary for orientation, but it must link to the canonical owner
instead of repeating volatile details.
Volatile details include commands, flags, configuration fields and defaults,
stage or integration keys, schemas, file names, paths, status codes, retry
behavior, and runtime guarantees. If readers could reasonably treat a statement
as a contract, maintain it only in the owning document.
### Current And Future Behavior
Outside `docs/roadmap/`, documentation describes implemented behavior only.
Partial features may be described only to their implemented boundary.
ADRs are the narrow exception: an ADR may record an accepted architectural
decision before implementation, but acceptance must not be presented as proof
that the behavior exists. The roadmap owns implementation status and sequencing
until the decision is implemented. Current architecture, user, operator,
integration, and internal documentation are updated when the behavior lands.
### Audience And Detail
Write for the document's stated audience and include only the detail needed for
its owned topic. User and operator docs should not expose implementation detail.
Developer docs should link to user-facing and external contracts rather than
restate them.
### Examples
Complete copyable files belong in `examples/`. Documentation may use the
smallest illustrative snippet needed to explain its owned topic, but should link
to maintained examples instead of embedding a second complete copy.
Examples must be valid, secret-free, and tested where practical. Commands and
configuration used in documentation should match the application.
### Security And Privacy
Documentation and examples must not contain real credentials, private keys,
private environment dumps, sensitive source material, or private infrastructure
details unless intentionally public. Document secret-handling mechanisms, not
secret values.
## Canonical Ownership
| Topic | Canonical owner | Owned content | Content owned elsewhere |
| --- | --- | --- | --- |
| Product orientation and minimal end-to-end quickstart | `README.md` | What Narratio is, why it is useful, one shortest successful invocation, and links onward. | Complete command reference, configuration reference, operational procedures, implementation detail. |
| Contributor entry point | `docs/development.md` | Task-oriented reading guide, minimal contributor orientation, baseline validation commands, and links to canonical docs. | Package inventory, architecture rules, subsystem behavior, detailed change recipes. |
| Current application architecture | `docs/policy/architecture.md` | System shape, normative ownership, dependency direction, architectural boundaries, invariants, safety properties, and non-goals. | Concrete package inventory, implementation mechanics, contributor procedures, decision history, future work. |
| Documentation organization | `docs/policy/documentation.md` | Documentation ownership, audience boundaries, maintenance rules, and ADR/document lifecycle. | Application architecture or product behavior. |
| Testing policy | `docs/policy/testing.md` | Test philosophy, risk-based sufficiency, test boundaries, doubles, coverage guidance, regression-test policy, and criteria for adding, rewriting, or deleting tests. | Subsystem behavior, application contracts, subsystem-specific test inventories, and implementation plans. |
| CLI contract | `docs/cli.md` | Commands, arguments, flags, invocation semantics, output conventions, and exit behavior. | End-to-end operating procedures, configuration field definitions, runtime filesystem layout, stage implementation details. |
| Configuration contract | `docs/config.md` | Discovery and precedence, file schemas, fields, defaults, environment overrides, validation rules, and user-selectable stage or integration settings. | Complete example files, CLI syntax, runtime state lifecycle, implementation details. |
| Operations | `docs/operations.md` | Runtime workflows, physical filesystem and remote-state layout, output and diagnostic handling, resume, cleanup, permissions, recovery, and operational limits. | CLI flag syntax, configuration field definitions, logical artifact schemas, implementation mechanics. |
| Troubleshooting | `docs/troubleshooting.md` | Symptom-driven diagnosis, likely causes, safe inspection steps and remedies, and links to relevant contracts. | CLI syntax, configuration definitions, operational procedures, integration contracts, implementation mechanics. |
| Public HTTP contract, if introduced | `docs/api.md` | Routes, authentication, media types, request and response schemas, status codes, pagination, caching, idempotency, rate limits, and HTTP retry semantics. | Client walkthroughs, upstream or downstream integration internals, implementation detail. |
| Consumer guidance, if a public package or API is introduced | `docs/consumers/` | Task-oriented use of the public interface, minimal client examples, and consumer responsibilities. | HTTP wire semantics, external protocol contracts, internal implementation detail. |
| External and durable integration contracts | `docs/integrations/` | External file formats and protocols, upstream and downstream contracts, logical artifact paths and schemas, media types, and compatibility behavior. | Physical runtime placement and lifecycle, internal transformations, CLI syntax, configuration defaults. |
| Implemented component inventory | `docs/internal/overview.md` | Current packages and components, their implemented responsibilities, and links to focused internal docs. | Normative architecture, contributor reading policy, external contracts. |
| Internal component behavior | Other files under `docs/internal/` | Implementation flow, internal collaborators and state transitions, package-local guarantees and failures, and relevant tests. | Global architecture invariants, configuration definitions and defaults, external schemas, operator procedures. |
| Architectural decision history | `docs/adr/` | Significant decisions, context, alternatives, rationale, consequences, and supersession history. | Current behavior reference, implementation status, task sequencing. |
| Future work and implementation status | `docs/roadmap/` | Proposed, accepted, deferred, or rejected work; implementation status; sequencing; and task breakdowns. | Implemented behavior reference and architectural decision rationale. |
| Complete copyable artifacts | `examples/` | Maintained configuration, inputs, and other files intended to be copied or run. | Field-by-field reference, command reference, prose explanation. |
Documents that do not exist are required only when the corresponding interface
or responsibility exists. Do not create placeholder API, consumer, integration,
or operations documents for behavior the application does not have.
## Boundary Rules
### Orientation
The README owns product orientation. The developer guide routes contributors.
Architecture owns normative structure. Internal overview owns the current
concrete component map. These documents may link to one another but should not
maintain parallel package or behavior descriptions.
### Commands, Configuration, Operations, And Troubleshooting
CLI documentation answers how to invoke the application. Configuration
documentation answers what settings mean. Operations answers what happens to
runtime state and how to operate or recover the application. Troubleshooting
starts from observable symptoms and links readers to the owning command,
configuration, operational, or integration contract. When a workflow crosses
these topics, choose the document that owns the task and link to the other
contracts.
### Contracts And Implementation
Integration and API documents define externally observable shapes and
semantics. Internal documents explain how Narratio implements or consumes those
contracts. Internal docs may name a field, file, or protocol to identify a
dependency, but must link to its canonical contract for the definition.
### Security Topics
This policy owns what documentation and examples may contain. Architecture owns
application security invariants. Configuration owns credential-supply
mechanisms. Operations owns permissions and handling of sensitive runtime
artifacts. Troubleshooting owns safe diagnostic and remediation guidance.
Internal docs own implementation mechanisms only.
## Architecture Decision Records
Use sequentially numbered ADR filenames such as
`0001-record-architecture-decisions.md`. Follow the lightweight Nygard format:
1. title;
2. status;
3. date;
4. context;
5. decision;
6. alternatives considered;
7. consequences.
Treat the decision content of an accepted ADR as immutable. When a decision
changes, create a new ADR and update the earlier ADR's status to superseded.
Rejected architectural alternatives belong in the ADR; rejected product ideas
belong in the roadmap.
## Maintenance
When behavior changes, update its canonical owner in the same change. If
ownership moves, remove the old definition and replace it with a link where
navigation remains useful.
Before completing documentation work:
- verify affected behavior and examples;
- check commands, flags, fields, defaults, schemas, and paths against their
implementation;
- keep unimplemented behavior in the roadmap, subject to the ADR exception;
- remove stale references and validate links;
- confirm that non-owning documents summarize and link rather than redefine;
- confirm that no secrets or sensitive private data were added.

296
docs/policy/testing.md Normal file
View File

@@ -0,0 +1,296 @@
# Testing Policy
## Purpose
Our tests exist to make **incorrect changes expensive and correct changes cheap**.
We do not optimize for test count, line coverage, exhaustive isolation, or the fewest possible tests. We optimize for sufficient confidence in important behavior while imposing as little unnecessary friction as possible on future development.
## Every test has a cost
Testing is not an unqualified good. Every test imposes both an immediate cost and a continuing lifetime cost.
A test must be:
- written and reviewed;
- understood by future maintainers and coding agents;
- executed in local and CI workflows;
- diagnosed when it fails;
- updated when legitimate behavior changes;
- maintained as fixtures, APIs, and dependencies evolve; and
- removed or rewritten when it becomes redundant, brittle, misleading, or obsolete.
Tests also create cognitive and architectural friction. They can constrain refactoring, duplicate policy, slow feedback loops, add noise to failures, and cause harmless implementation changes to require unrelated edits across the suite.
A test is warranted only when the confidence it provides justifies these costs.
Apply this cost-benefit analysis at two levels:
1. **Per test:** What realistic defect does this test detect, how consequential would that defect be, and is that protection worth the test's lifetime cost?
2. **Across the suite:** Does this collection provide materially more confidence than a smaller, simpler suite would?
The preferred test suite is a **lean suite that provides sufficient confidence in the risks that matter, without redundant or low-value tests**. We seek sufficient confidence with the least unnecessary testing friction, not the fewest possible tests.
Some friction is intentional. Tests should make dangerous changes—such as breaking compatibility, corrupting data, violating security boundaries, or reintroducing subtle bugs—require deliberate review. They should not make ordinary internal changes needlessly expensive.
The cost of a test is not a reason to omit testing by default. Do not cite maintenance cost abstractly. When omitting a plausible test, be able to state why the protected failure is low-risk, already covered, obvious, reversible, or cheaper to detect elsewhere. For consequential, subtle, or difficult-to-observe behavior, the presumption should favor testing.
## Default testing style
Use a **classical/Detroit-style** approach:
- Test observable behavior, resulting state, contracts, and invariants.
- Use real internal collaborators when they are fast and deterministic.
- Use fakes, stubs, or mocks primarily at expensive, nondeterministic, destructive, or external boundaries.
- Prefer package-level behavioral tests over tests coupled to private helpers or internal call sequences.
- Treat exact collaborator interactions as testable behavior only when the interaction itself is a requirement.
Examples of appropriate seams include clocks, randomness, subprocesses, remote APIs, object storage, email, and paid LLM calls.
## Test execution requirements
Tests in the default suite must be deterministic, offline, and independent of real credentials. They must not invoke paid APIs or depend on mutable external services. Tests that require live infrastructure must be explicitly opt-in and clearly separated from the default suite.
Control clocks, randomness, environment variables, and other process-global or machine-specific state when they affect behavior. Tests should be safe to run repeatedly and alongside other tests without depending on execution order or state left by an earlier test.
## What deserves tests
Prioritize tests for:
1. Public and package-level contracts.
2. Domain rules and important invariants.
3. Boundary conditions and malformed input.
4. Failure handling, cancellation, retries, recovery, and partial success.
5. Serialization, schemas, compatibility, and round trips.
6. Previously observed or plausible regressions.
7. Representative integration and end-to-end workflows.
A package-level contract is behavior relied upon by another package or major collaborator, not every observable detail of a package implementation.
For behavior involving **data integrity, destructive operations, compatibility, security, concurrency, idempotency, or recovery**, presume that durable tests are required unless the behavior is already credibly protected at another layer.
Do not add tests merely because a function, branch, or line exists. Do not add a test when the same meaningful risk is already adequately protected elsewhere.
## Choose the right test boundary
Test through the narrowest stable boundary that expresses the behavior clearly.
This is often the package API, but it may instead be:
- a smaller pure function when dense domain logic is most clearly isolated there;
- a package-level operation when several internal collaborators jointly produce the behavior; or
- a larger integration boundary when correctness emerges from interaction with a real dependency.
Do not force all behavior through oversized end-to-end tests. Do not test every private helper merely because it exists. Choose the boundary that gives durable confidence with the least incidental coupling.
## Test behavior, not implementation
A test should protect a decision, contract, or invariant—not memorialize the current implementation.
Before adding or retaining a test, ask:
> What realistic defect would this test catch?
A test is suspect when its main purpose is to detect that someone:
- changed an internal constant;
- renamed or split a private helper;
- reordered equivalent internal operations;
- changed incidental formatting;
- replaced one correct algorithm with another; or
- refactored internal object structure without changing behavior.
Refactoring should normally require no test edits unless the refactored structure is itself part of the contract.
A test can be factually correct and still have negative value. Accurately describing current behavior is not enough; the protected behavior must be important enough to justify the future friction.
## Expected effects of different changes
Use the following expectations when evaluating test failures and test maintenance:
| Change | Expected effect on tests |
|---|---|
| Internal refactor that preserves behavior | Existing tests should normally remain unchanged and continue to pass. |
| Change to an internal default with no contractual significance | Behavioral tests should normally remain unchanged; tests should derive expectations from configuration or relationships rather than duplicate the old value. |
| Intentional change to public behavior, policy, schema, or compatibility guarantees | The relevant tests should be reviewed and changed deliberately. |
| Accidental violation of a contract or invariant | Tests should fail; fix the production code rather than rewriting the tests to accept the defect. |
A test failing is not the same as a test needing to be edited. Many tests may correctly fail because of one production defect. The maintenance smell is a correct internal change that requires unrelated expectation updates throughout the suite.
## Separate mechanism from policy
Configurable thresholds and defaults must not be duplicated throughout the test suite.
For example, do not encode an internal concurrency limit indirectly:
```go
// Production policy:
const maxConcurrency = 4
// Brittle test:
err := startProcesses(5)
require.Error(t, err)
```
Instead, test the mechanism relationally:
```go
const limit = 2
runner := NewRunner(limit)
require.NoError(t, runner.Start(limit))
require.ErrorIs(t, runner.Start(limit+1), ErrTooMuchConcurrency)
```
The test should prove:
- the configured limit is accepted; and
- one beyond the configured limit is rejected.
The production default should be tested exactly only when its literal value is itself a public, operational, safety, protocol, or compatibility requirement.
Apply the same rule to limits, timeouts, capacities, retry counts, and ranges: test relationships and behavior, not duplicated literals.
For concurrency limits, test both kinds of behavior when relevant:
1. **Configuration enforcement:** invalid or excessive requested values are handled correctly.
2. **Runtime enforcement:** observed peak concurrency never exceeds the configured limit.
Use a test-controlled limit and measure the behavior relative to that limit. Do not merely assert today's default value.
## Avoid semantic duplication across layers
Each behavior should have a clear test owner.
- Parser tests own parsing cases.
- Validator tests own validation rules.
- Domain tests own transformations and invariants.
- Adapter tests own external integration behavior.
- Orchestrator tests own coordination and failure propagation.
- CLI tests own argument and configuration mapping.
- End-to-end tests prove that representative assembled workflows work.
Higher-level tests should not repeat every lower-level case. A single intentional policy change should not require unrelated edits across many test files.
Tests that are individually reasonable may still be collectively redundant. Evaluate the marginal value of each additional test in light of the protection already provided by the rest of the suite.
## Use test doubles deliberately
Choose the least elaborate test double that provides the required control or observation.
As a default:
1. Prefer real collaborators when they are fast and deterministic.
2. Use small in-memory fakes when realistic stateful behavior is helpful.
3. Use stubs when a dependency only needs to provide controlled responses.
4. Use mocks when the interaction itself is contractual.
Mocks are appropriate when the contract includes facts such as:
- a notification is sent exactly once;
- a transaction is committed only after successful writes;
- cancellation reaches a subprocess;
- an expensive API is called no more than once; or
- a security audit event is emitted.
Do not use mocks merely to isolate every object or reproduce the implementation's call graph.
## Go-specific guidance
Use:
- table-driven tests for meaningful behavioral categories and boundaries;
- `t.TempDir()` for real filesystem behavior;
- `httptest.Server` for realistic HTTP interactions;
- fuzz tests for parsers, normalization, path handling, and broad input spaces;
- golden files only when the complete output is intentionally stable;
- integration tests where correctness depends on component interaction; and
- a small number of representative end-to-end tests.
Avoid exact error-string assertions unless the wording is itself contractual. Prefer `errors.Is`, `errors.As`, typed errors, or structured error fields.
At CLI boundaries, prefer exit classifications, structured output, and the smallest stable semantic fragment needed to identify the error. Do not snapshot complete diagnostic wording unless it is contractual.
Golden-file updates must require an explicit local flag. CI must not update golden files automatically, and reviewers must inspect the semantic diff before accepting an update.
Keep tests readable and direct. Test helpers and fixture frameworks must earn their own maintenance cost; do not build elaborate test infrastructure for small or isolated needs.
## Coverage
Coverage is a diagnostic, not a target.
Use it to find untested critical branches and unexpectedly weak packages. Do not write low-value tests solely to increase a percentage, and do not infer test quality from coverage alone.
Pure domain logic will often warrant higher coverage than CLI wiring or external adapters. Uneven coverage is acceptable when it reflects risk.
Increasing coverage is valuable only when the newly covered behavior protects a meaningful risk at an acceptable cost.
## Regression tests
A bug fix should normally include a regression test that fails before the fix and passes afterward.
Retain the test when the defect could realistically recur and its consequences justify the ongoing cost. Prefer the narrowest durable test of the violated contract or invariant; do not preserve accidental implementation details from the original bug.
Not every historical bug requires a permanent test. If the underlying design has made recurrence impossible, the test has become redundant, or a stronger invariant test now subsumes it, remove or consolidate it.
## Deleting or rewriting tests
Tests are maintained code, not permanent historical artifacts.
Delete or rewrite a test when its maintenance cost exceeds the confidence it provides.
Strong candidates include tests that:
- require updates after harmless internal changes;
- directly assert private constants without protecting a real contract;
- duplicate the same policy across several layers;
- verify mock choreography rather than outcomes;
- snapshot large amounts of incidental output;
- test trivial private helpers already exercised through stable package behavior;
- protect risks already covered more effectively elsewhere;
- are flaky, misleading, obsolete, or disproportionately expensive to diagnose; or
- no longer correspond to a plausible failure mode.
Several brittle tests may encode one genuine requirement. Replace them with one durable behavior-level or invariant test rather than preserving all of them.
Deleting a low-value test can improve the quality of the suite by reducing noise, maintenance burden, and friction around legitimate change.
## Reviewing a proposed test
Use the following questions when the value, boundary, or durability of a proposed test is not self-evident. Significant test additions should be reviewable against them, but written answers are not required for every routine test.
1. What realistic defect would it catch?
2. How likely is that defect?
3. How consequential would it be?
4. Is the behavior already protected elsewhere?
5. At which layer should this behavior be owned?
6. Does the test assert a durable contract or an incidental implementation detail?
7. Could the implementation be refactored without changing the behavior and without editing this test?
8. What should cause this test to fail?
9. What legitimate changes should not cause this test to fail?
10. What ongoing maintenance, execution, and diagnostic cost will the test impose?
11. Is there a smaller or more direct test that protects the same risk?
Do not add the test when its expected lifetime cost exceeds its expected protective value.
When deciding not to test plausible behavior, record or be able to explain why the risk is low, already protected, obvious, reversible, or cheaper to detect elsewhere.
## Definition of sufficient
A test suite is sufficient when:
- important contracts and invariants are protected;
- meaningful boundaries and failure modes are exercised;
- realistic and consequential regressions are credibly protected against silent recurrence;
- behavior involving data integrity, destructive operations, compatibility, security, concurrency, idempotency, and recovery is credibly protected;
- important external boundaries have realistic integration coverage;
- representative complete workflows are tested;
- failures provide useful signal rather than redundant noise;
- legitimate internal changes usually do not require test edits; and
- additional tests would mostly repeat existing protection or preserve inconsequential implementation details.
Sufficiency is a risk judgment, not a coverage percentage or test count. Reassess it as the application, its users, and the consequences of failure evolve.
The governing rule is:
> Test heavily where failure is consequential, subtle, or difficult to detect after the fact. Test lightly where failure is obvious, reversible, and inexpensive—and retain no test whose lifetime cost exceeds the confidence it provides.

View File

@@ -1,715 +0,0 @@
# Roadmap: `narratio restore` Subcommand
## Status
Implemented through Step 8. This document remains as roadmap and design history for the restore feature, and as the home for future restore-related ideas (for example `run --restore`).
## Summary
Add a new `narratio restore` subcommand that hydrates a local session workspace from the current committed remote archive state.
The primary operator workflow is:
```bash
narratio restore --session-id 2026-04-04
narratio run-stage --force analyze
```
This should allow a new machine with no local workspace state to restore the durable session manifest, transcripts, and generated artifacts from S3, then generate new Scriptorium artifacts without re-running transcription, merge, normalize, polish, or trim.
This is intentionally a separate command. Do not fold this behavior into the `prepare` stage. The existing `prepare` stage should remain focused on materializing configured local/S3 inputs for a pipeline run.
## Goals
- Add a first-class `narratio restore` command.
- Restore the current committed remote session state into the canonical local session workspace.
- Use the existing object storage adapter boundary.
- Preserve archive commit semantics: only restore from a remote state that has a valid current commit marker.
- Restore durable session-level outputs needed for downstream stages, especially `analyze`.
- Provide safe conflict behavior by default.
- Support `--dry-run`, `--force`, and `--include-audio`.
- Keep the implementation explicit, testable, and narrow.
## Non-goals
- Do not make `restore` a pipeline stage.
- Do not change the `prepare` stage behavior as part of this work.
- Do not add implicit restore behavior to `narratio run` in this implementation.
- Do not restore historical run-local sandboxes by default.
- Do not implement a generic remote synchronization engine.
- Do not implement bidirectional sync.
- Do not delete local files merely because they are absent remotely.
- Do not merge remote and local manifests in the first implementation.
- Do not require live S3 for the ordinary unit test suite.
## Future work explicitly out of scope
A future change may add:
```bash
narratio run --restore
```
That future flag should run `narratio restore` before starting the normal pipeline. Mention this as future work in roadmap/docs if useful, but do not implement it now.
## Existing architecture to preserve
### `prepare` remains input materialization
The `prepare` stage currently materializes required session inputs into canonical local workspace paths and records input provenance. It owns local copying/materialization of config and audio inputs, including S3 audio download when `session.inputs.audio_s3.prefix` is configured. It does not own transcript generation/processing or archive publish behavior.
`restore` should not be implemented by expanding `prepare`. It should be an app-level command that reuses shared helpers where appropriate.
### Workspace model
The local durable session workspace is campaign-aware:
```text
{workspace.root}/work/{campaign}/{session_id}/
```
It contains durable session paths such as:
```text
manifest.json
inputs/
audio/
transcripts/
artifacts/
reports/
logs/
config/
current/
runs/
```
Run-local sandboxes live below:
```text
runs/{run_id}/
```
Restore should target durable session-level paths, not old run-local stage sandboxes.
### Storage boundary
The storage adapter owns object-store primitives only: `List`, `Download`, `Upload`, and `Exists`.
The storage adapter must not infer root prefixes, campaign names, session IDs, run IDs, or archive layout. Restore code must construct full bucket-relative keys before calling storage.
### Archive commit boundary
A remote run is current only after the archive stage has uploaded the run record, promoted outputs, `current/manifest.json`, and finally `current/run_id.txt`.
`current/run_id.txt` is the final remote commit marker and must be written last.
Restore must not treat incomplete, skipped, failed, or uncommitted archive attempts as current remote state.
## User-facing command
Add:
```bash
narratio restore [flags]
```
The command should use the same configuration/session discovery conventions as `run`, `plan`, `resume`, and `run-stage` where practical:
```bash
narratio restore --config /path/to/pipeline.yml --session ./session.yml --session-id 2026-04-04
```
Required effective inputs:
- resolved pipeline config;
- resolved session config;
- `session.campaign`;
- `session.session_id`;
- configured remote storage backend.
Supported flags:
```text
--config <path> Existing pipeline config path behavior.
--session <path> Existing session config path behavior.
--session-id <value> Existing session template behavior.
--dry-run Plan restore actions without writing local files.
--force Overwrite conflicting local files with remote files.
--include-audio Include archived session-level audio files.
```
Do not add `--restore` to `run` in this implementation.
## Default restore scope
By default, restore:
1. Validates and reads the current remote commit marker.
2. Downloads the current remote manifest into the local session manifest path.
3. Downloads durable transcript files.
4. Downloads durable generated artifact files.
Default included remote/local durable paths:
```text
manifest.json from remote current manifest
transcripts/**
artifacts/**
```
Default excluded paths:
```text
audio/** unless --include-audio is passed
runs/** always excluded for this implementation
logs/** excluded for this implementation
reports/** excluded for this implementation unless needed for current manifest validation
config/** excluded for this implementation
inputs/** excluded for this implementation
current/** remote control metadata only; do not mirror blindly
```
If the existing archive implementation stores promoted files in a different remote layout, use the existing archive/path helpers and current archive semantics rather than inventing a parallel layout.
## Remote state discovery
Implement restore around the current committed archive state.
Expected algorithm:
1. Resolve pipeline/session config.
2. Ensure storage is configured.
3. Ensure local workspace layout exists.
4. Acquire the session lock.
5. Build the remote session archive prefix using the same helpers/policy used by archive code.
6. Check for the remote `current/run_id.txt` commit marker.
7. Read the committed run ID.
8. Download `current/manifest.json` to a temporary file.
9. Validate that the manifest is parseable and belongs to the requested campaign/session.
10. Build a restore plan from the committed remote state.
11. Execute the restore plan unless `--dry-run` is set.
12. Emit a concise summary.
Important: `current/run_id.txt` is the commit marker. Do not restore from a remote session prefix merely because files exist under `transcripts/` or `artifacts/`.
## Restore planning
Create a planning layer before writing files.
A restore plan entry should include at least:
```go
type RestoreAction struct {
Kind RestoreActionKind
RemoteKey string
LocalPath string
Size int64
ETag string
ExistsLocal bool
SameLocal bool
Conflict bool
Reason string
}
```
Suggested action kinds:
```text
download
skip_same
skip_missing_optional
conflict
```
The restore planner should be deterministic:
- sort remote objects by key;
- sort planned actions by local path or stable restore priority;
- write/report stable output for tests.
## Conflict and overwrite policy
Default behavior should be safe.
For each planned file:
```text
local absent:
download
local present and same as remote:
skip
local present and different:
conflict; fail restore unless --force is set
--force:
overwrite local conflicting files with remote versions
--dry-run:
do not write any files; report what would happen
```
The first implementation may use size and checksum/hash comparison where available. If remote ETag cannot be treated as a content hash, compare by downloading to a temporary file and hashing locally before deciding whether a local file is the same. Prefer correctness over assuming provider-specific ETag semantics.
Do not delete local files that are not present remotely.
## File writing and transactionality
Restore should avoid partial writes.
Implementation requirements:
- download each remote object to a temporary file under the session workspace or OS temp dir;
- validate downloaded content where possible before replacing local files;
- create parent directories as needed;
- atomically rename/copy into place only after successful download;
- do not overwrite local files unless `--force` is set;
- if a later file fails, preserve already-restored files but return a failure summary;
- never corrupt an existing local manifest on failed manifest download/parse.
Manifest restore is especially sensitive:
- download remote `current/manifest.json` to a temporary file;
- parse and validate it;
- if no local manifest exists, install it;
- if a local manifest exists and is equivalent, skip;
- if a local manifest exists and differs, fail unless `--force` is set;
- with `--force`, replace the local manifest with the remote manifest after validation;
- do not attempt a manifest merge in the initial implementation.
## Manifest semantics
`restore` is not a pipeline run and should not mark stages as running/succeeded/failed.
The restored remote manifest becomes the local session manifest. That is what allows a subsequent command such as:
```bash
narratio run-stage --force analyze
```
to see existing upstream stage state and canonical durable outputs.
Do not create a new run manifest for `restore`.
It is acceptable to write a restore diagnostic report outside the manifest, for example:
```text
reports/restore-latest.json
```
or a timestamped report, if that pattern fits the existing codebase. The report must not contain secrets.
## Local workspace locking
`restore` should acquire the same session lock used by ordinary pipeline operations before modifying session workspace state.
If the lock is held, fail fast with the same lock-conflict behavior used elsewhere.
`--dry-run` may still acquire the lock for consistency, but it is acceptable to avoid the lock if the codebase already has a clear read-only command pattern. Prefer safety and simplicity.
## Audio behavior
By default, do not restore audio.
If `--include-audio` is passed:
- restore archived durable session-level audio files only;
- do not use run-scoped spool paths;
- do not mutate or delete spool state;
- do not infer original `session.inputs.audio_s3.prefix` behavior;
- respect the same conflict/force/dry-run behavior used for transcripts/artifacts.
If the archive does not contain durable audio files, `--include-audio` should report that no archived audio was found rather than failing, unless the final implementation chooses to treat explicit audio restore as required. Prefer non-failure for absent archived audio unless tests or existing archive semantics suggest otherwise.
## Remote object selection
Prefer using manifest/artifact metadata when it reliably identifies durable outputs.
Also support listing committed durable archive prefixes so restore can retrieve all top-level session artifacts that may not yet be fully represented in manifest metadata.
The implementation should inspect existing archive code before choosing the final object-selection method. Do not duplicate archive path construction.
Recommended selection priority:
1. Remote current manifest path.
2. Durable promoted transcript/artifact outputs recorded in the manifest or archive metadata, if available.
3. Objects under committed durable `transcripts/` and `artifacts/` archive prefixes.
4. Objects under durable `audio/` only when `--include-audio` is passed.
Always exclude:
```text
runs/**
```
for the first implementation.
## Package and file organization
Expected areas to inspect and update:
```text
cmd/narratio/
internal/app/
internal/adapters/storage/
internal/artifacts/
internal/manifest/
docs/
examples/
```
Suggested implementation shape:
```text
internal/app/restore.go
internal/app/restore_test.go
internal/archive/restore/
planner.go
executor.go
report.go
keys.go
*_test.go
```
The exact package name may vary. Use whatever best fits the existing repository, but keep these boundaries clear:
- `internal/app` owns CLI command handling, config/session loading, lock acquisition, and wiring.
- Restore planning/execution owns remote key discovery, conflict detection, downloads, and reporting.
- `internal/adapters/storage` remains a transport boundary only.
- Workspace/path helpers remain centralized; do not scatter string concatenation.
If the repository already has an `internal/archive` or archive-stage helper package, prefer extending that rather than creating a conflicting package layout.
## CLI output
`narratio restore` should print a concise operator summary.
Example successful output:
```text
Restored session archive for sample-campaign/2026-04-04
Remote run: 20260504T031500Z-a1b2c3
Downloaded: 4
Skipped unchanged: 2
Conflicts: 0
```
Example dry run:
```text
Restore plan for sample-campaign/2026-04-04
Remote run: 20260504T031500Z-a1b2c3
Would download: transcripts/processed.json
Would download: transcripts/trimmed.json
Would skip unchanged: artifacts/session_recap.md
```
Example conflict:
```text
restore conflict: local artifacts/session_recap.md differs from remote archive; rerun with --force to overwrite
```
Do not print transcript or artifact content.
## Error behavior
Fail clearly when:
- storage backend is not configured;
- S3 bucket/config is missing or invalid;
- remote current commit marker is missing;
- remote current manifest is missing;
- remote manifest is invalid;
- remote manifest does not match requested campaign/session;
- local file differs from remote and `--force` is not set;
- a required remote object download fails;
- a local path would escape the session workspace;
- a remote key maps to an unsafe local path.
Skip or report non-fatal conditions when:
- optional audio restore finds no archived audio;
- an included prefix has no objects;
- a local file already matches the remote file.
## Path safety
Every restored file must map to a safe path under the session root.
Validation rules:
- local restore paths must be relative to the session root;
- reject absolute paths;
- reject `..` traversal;
- reject paths that escape through symlinks if the codebase has symlink-safe path checks;
- do not restore remote keys directly without mapping/classification;
- do not mirror arbitrary remote keys.
## Testing plan
Add focused unit tests. Do not require live S3.
### CLI tests
Add or update `internal/app` command tests for:
- `narratio restore --help`;
- restore accepts `--config`, `--session`, and `--session-id`;
- restore accepts `--dry-run`;
- restore accepts `--force`;
- restore accepts `--include-audio`;
- restore fails when storage is not configured;
- restore does not run pipeline stages.
### Restore planner tests
Test:
- missing `current/run_id.txt` fails;
- missing `current/manifest.json` fails;
- invalid manifest fails;
- wrong campaign/session manifest fails;
- default scope includes manifest/transcripts/artifacts;
- default scope excludes audio/logs/reports/config/runs;
- `--include-audio` includes durable audio;
- run-local keys are excluded;
- keys are sorted deterministically;
- unsafe remote-to-local paths are rejected.
### Conflict policy tests
Test:
- absent local file downloads;
- matching local file skips;
- differing local file conflicts by default;
- `--force` overwrites conflicts;
- `--dry-run` writes nothing;
- partial failure does not corrupt an existing local manifest.
### Storage/fake tests
Use fake storage to simulate:
- object listing;
- object download;
- missing objects;
- download failures;
- metadata/ETag behavior.
### Workspace/lock tests
Test:
- session layout is created before restore;
- session lock conflict fails;
- restored files land under the expected campaign/session workspace;
- no files are written outside the session root.
### Follow-up command workflow test
Add at least one test that simulates:
```bash
narratio restore --session-id 2026-04-04
narratio run-stage --force analyze
```
The test does not need to run real Scriptorium. Use existing fake/stub behavior to verify that restored transcripts and manifest state are sufficient for analyze-stage input resolution.
## Documentation updates when implemented
When the feature is implemented, update current-behavior docs:
```text
docs/cli.md
docs/operations.md
docs/internal/storage.md or docs/internal/archive/restore.md
```
If the documentation set does not yet have an internal restore document, add one consistent with the existing internal-doc style:
```text
docs/internal/command-restore.md
```
or:
```text
docs/internal/archive-restore.md
```
Do not document future `narratio run --restore` behavior outside `docs/roadmap/` until implemented.
## Implementation phases
### Phase 1: Audit existing archive and path helpers (completed)
Before coding behavior, inspect:
```text
internal/app/
internal/stage/archive*
internal/adapters/storage/
internal/artifacts/
internal/manifest/
docs/internal/stage-archive.md, if present
```
Determine:
- exact remote archive key layout;
- how root prefix/campaign/session are modeled;
- how current commit marker keys are built;
- how current manifest is uploaded;
- where promoted outputs are uploaded;
- whether helper functions already exist for remote archive keys;
- whether local workspace path helpers can safely map restore destinations.
Deliverable:
- small code comments or internal helper selection;
- no large behavior change yet unless required by tests.
### Phase 2: Add CLI surface and command wiring (completed)
Add `narratio restore` command parsing.
Wire flags:
```text
--config
--session
--session-id
--dry-run
--force
--include-audio
```
Use the existing config/session load path where practical.
Deliverable:
- command exists;
- help output is sensible;
- command validates basic inputs;
- command returns a clear “not yet implemented” or calls an empty planner if phased commits are desired;
- CLI tests pass.
### Phase 3: Implement remote current-state discovery (completed)
Add restore code that:
- creates an object store from resolved config;
- builds remote current marker key;
- reads `current/run_id.txt`;
- reads/downloads `current/manifest.json`;
- validates manifest identity;
- returns remote current-state metadata.
Deliverable:
- fake-storage tests for current-state discovery;
- no local file writes beyond temporary files.
### Phase 4: Implement restore planning (completed)
Build deterministic restore plans for default scope and `--include-audio`.
Deliverable:
- plan lists manifest, transcript, artifact files;
- plan excludes run-local data;
- plan detects local same/conflict/missing states;
- dry-run output works;
- no real file overwrite yet except temp comparisons as needed.
### Phase 5: Implement restore execution (completed)
Execute the plan safely:
- create directories;
- download to temporary files;
- validate content where practical;
- atomically install files;
- enforce default conflict failure;
- support `--force`;
- preserve existing manifest unless safe to replace.
Deliverable:
- restore works end-to-end against fake storage;
- failures are clear and do not corrupt existing local manifest.
### Phase 6: Add restore report and operator summary (completed)
Add concise stdout summary and optional JSON restore report if consistent with project diagnostics.
Deliverable:
- user-friendly output;
- durable diagnostic report if implemented;
- no content leakage.
### Phase 7: Workflow integration test (completed)
Add a test for restoring a previous session and then forcing `analyze`.
Deliverable:
- restored manifest/transcripts/artifacts are sufficient for analyze input resolution;
- no upstream stages rerun;
- no reliance on live subprocesses or S3.
### Phase 8: Documentation update (completed)
Once implemented, update current-behavior docs and internal command docs.
Also leave future `narratio run --restore` in roadmap only.
## Definition of done
The feature is complete when:
- `narratio restore` exists and is documented.
- It uses the same config/session discovery semantics as other commands where practical.
- It requires configured remote storage.
- It restores only from a committed current archive state.
- It restores the current manifest, transcripts, and artifacts by default.
- It restores audio only with `--include-audio`.
- It excludes run-local sandboxes.
- It fails on local/remote conflicts by default.
- `--force` overwrites conflicts.
- `--dry-run` writes nothing.
- It uses fake storage in tests.
- It does not change `prepare` behavior.
- It does not implement `narratio run --restore`.
- It avoids AWS SDK leakage outside the storage adapter.
- It uses centralized path/key helpers rather than scattered string concatenation.
- `go test ./...` passes.
## Suggested test commands
Run focused tests first:
```bash
go test ./internal/app -run TestExecute -v
go test ./internal/adapters/storage -v
go test ./internal/artifacts -v
go test ./internal/manifest -v
```
Then run the full suite:
```bash
go test ./...
```
## Suggested commit message
```text
Add restore subcommand roadmap
```

View File

@@ -1,380 +1,479 @@
# Troubleshooting
## Purpose
Canonical operator troubleshooting guide for recurring implemented Narratio failures.
Operational diagnosis guide for common Narratio failures.
## Config file discovery failure
## Config file not found
Symptom:
- `run`, `plan`, `resume`, `run-stage`, or `restore` fails with config/session not found.
Likely Cause:
- `pipeline.yml` or `session.yml` is missing from discovery paths.
- wrong working directory when relying on `./session.yml`.
- command fails to resolve `pipeline.yml`, `campaign.yml`, or `session.yml`.
Likely causes:
- missing files in default search paths;
- wrong campaign selection;
- omitted explicit flags.
Diagnostics:
```bash
pwd
ls -l ./session.yml
ls -l /usr/local/etc/narratio/pipeline.yml /etc/narratio/pipeline.yml
narratio session plan 2026-04-04
```
Safe Fix:
- pass explicit `--config` and `--session`.
- or place files in documented discovery paths.
Safe fix:
Links:
- [docs/config.md](./config.md)
- [docs/cli.md](./cli.md)
- pass explicit `--config`, `--campaign` or `--campaign-file`, and `--session`.
## Session template rendering failure
Relevant reference: [Configuration discovery](./config.md#discovery-and-selection).
## Session template placeholders rejected
Symptom:
- load fails with unresolved placeholder or `session_id` mismatch.
Likely Cause:
- templated `session.yml` used without `--session-id`.
- rendered `session_id` differs from passed `--session-id`.
- load error says session file must be concrete or contains `{{ ... }}` placeholders.
Likely cause:
- using template content as runtime session config.
Diagnostics:
```bash
narratio plan --session ./session.yml --session-id 2026-04-04
narratio session validate 2026-04-04 --session /path/session.yml
```
Safe Fix:
- pass `--session-id` when template placeholders are present.
- ensure rendered `session_id` matches intended run session id.
Safe fix:
Links:
- [docs/config.md](./config.md)
- generate concrete session YAML with `narratio session init`.
## Strict YAML decode or validation failure
Relevant reference: [Operations: Session Initialization](./operations.md#session-initialization).
## Strict decode or schema validation failure
Symptom:
- config load fails with unknown field or validation error.
Likely Cause:
- typo/stale field name.
- missing required fields or invalid constraints.
- unknown field / invalid value error during config load.
Likely cause:
- stale field name, typo, invalid enum, or invalid duration/path format.
Diagnostics:
```bash
narratio plan --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04
narratio session plan 2026-04-04 --config /path/pipeline.yml --campaign-file /path/campaign.yml --session /path/session.yml
```
Safe Fix:
- align fields/values to canonical config reference and examples.
Safe fix:
Links:
- [docs/config.md](./config.md)
- [examples/](../examples/)
- align config with [Configuration](./config.md) and the
[maintained examples](../examples/README.md).
## `--artifacts` selection failure
Relevant reference: [Configuration](./config.md).
## Audio mode conflict
Symptom:
- `run`/`resume`/`run-stage` fails with invalid or unknown artifact selection.
Likely Cause:
- `--artifacts` contains blank names or unknown artifact keys.
- `pipeline.scriptorium.artifacts` missing while using `--artifacts`.
- validation fails on session audio configuration.
Likely cause:
- configured both local and S3 session audio inputs.
Diagnostics:
```bash
narratio run --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --artifacts player_handout
narratio session validate 2026-04-04
```
Safe Fix:
- use configured artifact keys only.
- ensure `pipeline.scriptorium.artifacts` is defined.
Safe fix:
Links:
- [docs/cli.md](./cli.md)
- [docs/config.md](./config.md)
- use local mode (`audio_dir` or `audio_files`) or S3 mode (`audio_s3.prefix`), not both.
## `run-stage --artifacts` on non-analyze stage
Relevant reference: [Session configuration](./config.md#session).
## `--artifacts` selection error
Symptom:
- `run-stage` fails with `--artifacts is only supported for stage "analyze"`.
Likely Cause:
- `--artifacts` was used with a non-`analyze` stage.
- unknown artifact key or invalid `--artifacts` usage.
Likely causes:
- key not defined in `pipeline.scriptorium.artifacts`;
- empty list entry (for example trailing comma);
- `run-stage` used with non-`analyze`/`publish` target.
Diagnostics:
```bash
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --artifacts session_recap polish
narratio session artifacts 2026-04-04
```
Safe Fix:
- use `--artifacts` only with `run-stage ... analyze`.
Safe fix:
Links:
- [docs/cli.md](./cli.md)
- provide only configured keys and use `--artifacts` with supported commands/stages.
## Configured artifact dependency/input validation failure
Relevant reference: [CLI artifact selection](./cli.md).
## Notarius executable missing
Symptom:
- config validation fails for `depends_on`, `narratio.artifact.<name>` source, or artifact output path.
Likely Cause:
- `narratio.artifact.<name>` source missing matching `depends_on` key.
- dependency references unknown artifact key.
- dependency self-reference or enabled dependency cycle.
- artifact output path missing/invalid/outside `artifacts/` root.
- extraction fails while resolving or starting the Notarius executable.
Likely causes:
- `pipeline.notarius.binary` is not installed, executable, or on `PATH`;
- a configured executable path is wrong.
Safe fix:
- install a compatible Notarius release or correct the binary setting, then
rerun extraction.
Relevant references: [Notarius configuration](./config.md#notarius-output-entries)
and [Notarius integration](./integrations/notarius.md).
## Notarius exits nonzero
Symptom:
- extraction reports a Notarius exit error instead of a receipt.
Diagnostics:
- inspect `runs/{run_id}/extract/notarius.stderr.log`; stdout is reserved for
the receipt and is not merged with diagnostics.
Safe fix:
- correct the reported Notarius pipeline, input, provider, or configuration
failure and rerun extraction. Do not edit a staged output bundle into place.
After a failed replacement, an older immutable bundle may still exist even
though the current session manifest has no successful extraction payload. This
is expected audit state, not a signal to relink the old bundle manually.
Relevant reference: [Operations: Extraction Workflow](./operations.md#extraction-workflow).
## Atomic Notarius promotion unsupported
Symptom:
- extraction fails with `atomic no-replace directory promotion is unsupported`
before a durable bundle or temporary promotion tree is created.
Likely cause:
- Narratio is running on an operating system other than Linux, macOS, or
Windows, where the required atomic no-replace directory primitive has not
been implemented and verified.
Safe fix:
- run extraction on Linux, macOS, or Windows. Do not replace the atomic commit
with a manual copy or move; the session manifest must never observe a partial
or overwritten bundle.
This is an extraction-specific platform boundary, not a support statement for
unrelated Narratio workflows. See
[Operations: Extraction Workflow](./operations.md#extraction-workflow).
## Notarius receipt or index incompatible
Symptom:
- extraction rejects the receipt schema, pipeline identity, bundle/index path,
lane descriptor, or payload path even though Notarius exited successfully.
Likely causes:
- Narratio and Notarius versions disagree on their consumer contract;
- the configured pipeline or lane constraints are stale;
- output paths escape the bundle or traverse symlinks.
Safe fix:
- compare installed Notarius output with the canonical Notarius contracts,
including receipt `index_file: index.json` and index management names
`manifest.json`, `rejected.json`, and `warnings.json`; align
`pipeline.notarius` constraints and rerun. Do not bypass confinement or schema
checks.
Relevant reference: [Notarius integration](./integrations/notarius.md).
## Required Notarius lane rejected or missing
Symptom:
- extraction fails because a configured lane is rejected, missing, duplicated,
or incompatible, including after a zero exit.
Safe fix:
- inspect the Notarius diagnostic log and bundle rejection/warning information;
- correct the Notarius module or the exact declared lane contract;
- remove an output declaration only if downstream consumers genuinely no longer
require that source, then rerun extraction.
Every configured output is required. Narratio does not promote a partial result.
## Extraction resume invalidated
Symptom:
- a previously successful extraction runs again during ordinary continuation.
Likely causes:
- the executable/config path, pipeline ID, timeout, working directory, or
configured output contracts changed;
- the durable bundle, index, lane set, provenance, regular-file status, or
checksum no longer validates.
Safe fix:
- allow the automatic rerun after verifying the current configuration. Treat
an unsafe path or symlink error as filesystem corruption or tampering and
investigate it rather than replacing files manually.
## Notarius transitive configuration changed
Symptom:
- Notarius profiles, prompts, modules, imported files, or references changed,
but Narratio still considers the previous extraction resumable.
Safe fix:
```bash
narratio run-stage extract 2026-04-04 --force
```
Narratio fingerprints its invocation contract, not the contents of transitive
Notarius inputs. Always force extraction after changing them; downstream
successful stages are then marked stale normally.
Relevant reference: [Operations: Extraction Workflow](./operations.md#extraction-workflow).
## Previous-session artifact input missing
Symptom:
- prepare/analyze fails due to missing required previous-session artifact cache input.
Likely causes:
- missing `session.previous_session_id`;
- previous artifact not restored/published for source session.
Diagnostics:
```bash
narratio plan --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04
narratio session validate 2026-04-04
narratio session status 2026-04-04
```
Safe Fix:
- ensure artifact-to-artifact inputs have explicit `depends_on` entries using artifact keys.
- ensure referenced artifacts exist and define valid `output_path` values.
- keep output paths relative and under `artifacts/`.
Links:
- [docs/config.md](./config.md)
- [docs/internal/stage-analyze.md](./internal/stage-analyze.md)
## Required configured artifact input unavailable at analyze time
Symptom:
- analyze fails because configured input source is unavailable.
Likely Cause:
- required upstream configured artifact was not selected/executed this run.
- non-executable dependency output file is missing or invalid on disk.
Diagnostics:
Safe fix:
```bash
narratio status --manifest /path/to/manifest.json
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --artifacts player_handout analyze
narratio session restore 2026-04-04
```
Safe Fix:
- run analyze with needed artifacts selected.
- or ensure dependency output file exists at configured path and is valid.
Links:
- [docs/operations.md](./operations.md)
- [docs/config.md](./config.md)
## Manifest/status path failure
Symptom:
- `status` fails because manifest path is missing, unreadable, or invalid.
Likely Cause:
- wrong manifest path.
- manifest removed after cleanup.
- `--manifest` omitted.
Diagnostics:
or rerun prepare after correcting session config:
```bash
narratio status --manifest /path/to/manifest.json
ls -l /path/to/manifest.json
narratio run-stage prepare 2026-04-04 --force
```
Safe Fix:
- use manifest path printed by `run`, `resume`, or `run-stage`.
Links:
- [docs/cli.md](./cli.md)
- [docs/operations.md](./operations.md)
Relevant reference: [Operations: Restore Workflow](./operations.md#restore-workflow).
## Session lock conflict (`.lock`)
Symptom:
- `run`, `resume`, `run-stage`, or `restore` fails with lock conflict for session workdir.
Likely Cause:
- another Narratio process is running same session.
- stale lock from interrupted prior run.
- command fails acquiring session lock.
Likely causes:
- another process is running for the same session;
- stale lock left by interrupted process.
Diagnostics:
```bash
ls -l {workspace.root}/work/{campaign}/{session_id}/.lock
cat {workspace.root}/work/{campaign}/{session_id}/.lock
ps aux | grep narratio
```
Safe Fix:
- wait for active process to finish.
- if no process is active, remove only stale session `.lock` file.
Safe fix:
Links:
- [docs/operations.md](./operations.md)
- [docs/internal/workspace.md](./internal/workspace.md)
- wait for active process completion;
- remove stale lock only after confirming no live process owns it.
## Restore remote current pointer or manifest missing
Symptom:
- `restore` fails with remote current pointer or current manifest errors.
Likely Cause:
- `current/run_id.txt` was never published.
- `current/manifest.json` is missing for the session prefix.
- archive commit did not complete.
Diagnostics:
```bash
narratio restore --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --dry-run
```
Safe Fix:
- verify archive stage succeeded for the target session.
- rerun/archive from a healthy source workspace so current pointers are published.
Links:
- [docs/operations.md](./operations.md)
- [docs/internal/stage-archive.md](./internal/stage-archive.md)
## Restore manifest identity mismatch
Symptom:
- `restore` fails because remote manifest session or campaign does not match requested values.
Likely Cause:
- wrong `--session-id` or wrong session config selected.
- archive prefix points to a different campaign/session.
Diagnostics:
```bash
narratio restore --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --dry-run
```
Safe Fix:
- use the correct session config and `--session-id`.
- verify campaign/session identity in local config before restore.
Links:
- [docs/config.md](./config.md)
- [docs/operations.md](./operations.md)
Relevant reference: [Operations: Local State Layout](./operations.md#local-state-layout).
## Restore conflict without `--force`
Symptom:
- `restore` fails with `restore conflict` and conflict counts.
Likely Cause:
- local durable file differs from remote file for one or more planned restore paths.
- restore fails with conflict count.
Likely cause:
- local durable files differ from remote restore sources.
Diagnostics:
```bash
narratio restore --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --dry-run
narratio session restore 2026-04-04 --dry-run
```
Safe Fix:
- review planned conflicts.
- rerun with `--force` only when remote state should overwrite local state.
Safe fix:
Links:
- [docs/cli.md](./cli.md)
- [docs/operations.md](./operations.md)
- review conflicts;
- rerun with `--force` only when remote state should overwrite local.
## Restore report expectations
Relevant reference: [Operations: Restore Workflow](./operations.md#restore-workflow).
## Restore current-state discovery failure
Symptom:
- operator expects restore report file but does not find one.
Likely Cause:
- restore was executed in `--dry-run` mode.
- restore failed before report persistence path (for example lock acquisition failure).
- restore cannot find current pointer or current manifest.
Likely causes:
- no committed publish current state;
- storage credentials or connectivity failure.
Diagnostics:
```bash
ls -l {workspace.root}/work/{campaign}/{session_id}/reports/restore-latest.json
narratio session status 2026-04-04
narratio session restore 2026-04-04 --dry-run
```
Safe Fix:
- run non-dry-run restore for durable report output.
- resolve lock or early preflight failures and retry.
Safe fix:
Links:
- [docs/operations.md](./operations.md)
- resolve storage/auth issue;
- republish from healthy local state if current pointer is missing.
## Secrets env-dir or credential-env failure
Relevant reference: [Operations: Publish Workflow](./operations.md#publish-workflow).
## Publish output failure
Symptom:
- startup fails loading secrets directory, or stage fails due to missing credential env vars.
Likely Cause:
- invalid `pipeline.secrets.env_dir` path/permissions.
- required credential env var unset/empty.
- publish fails on missing required source, upload error, or commit write.
Likely causes:
- required source file not produced;
- lock/state expectations mismatch;
- remote storage failure.
Diagnostics:
```bash
narratio session artifacts 2026-04-04 --remote
narratio session status 2026-04-04
narratio run-stage publish 2026-04-04 --force
```
Safe fix:
- regenerate missing sources by rerunning prerequisite stages;
- correct publish source/destination rules;
- retry after storage failure is resolved.
Relevant reference: [Publish configuration](./config.md#publish-configuration-summary).
## Render markdown source missing
Symptom:
- analyze or publish fails because `narratio.transcript.final_markdown` or `narratio.transcript.final_trimmed_markdown` is unavailable.
Likely causes:
- render stage was not executed after transcript changes;
- render stage failed before producing canonical markdown outputs.
Diagnostics:
```bash
narratio session status 2026-04-04
```
Safe fix:
- rerun render and then retry downstream stage(s):
```bash
narratio run-stage render 2026-04-04 --force
narratio run-stage analyze 2026-04-04 --force
```
Relevant reference: [Operations: Stage Execution](./operations.md#stage-execution-and-continuation-behavior).
## Secrets or storage credential failure
Symptom:
- object-store command fails at initialization/auth.
Likely causes:
- invalid `pipeline.secrets.env_dir`;
- missing credential environment variables;
- invalid S3 endpoint/bucket settings.
Diagnostics:
```bash
ls -la /path/to/secrets_dir
env | grep -E 'AUDITA|OBJECT_STORAGE|AWS|SCRIPTORIUM'
env | grep -E 'OBJECT_STORAGE|AWS|AUDITA|SCRIPTORIUM'
```
Safe Fix:
- fix secrets directory and credential env vars.
Safe fix:
- correct secret-file path and permissions;
- provide required env vars;
- keep secret values out of YAML.
Links:
- [docs/config.md](./config.md)
Relevant reference: [Secrets](./config.md#secrets-handling).
## S3-audio prepare failure
## S3 audio prepare failure
Symptom:
- `prepare` fails in S3 mode (listing/downloading/no audio/backend error).
Likely Cause:
- wrong `session.inputs.audio_s3.prefix`.
- no `.flac` files at resolved prefix.
- invalid/missing object-store credentials or backend config.
- mixed local+S3 audio input config.
- prepare fails listing/downloading session S3 audio.
Likely causes:
- incorrect `session.inputs.audio_s3.prefix`;
- no matching `.flac` objects;
- storage connectivity or permissions failure.
Diagnostics:
```bash
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 prepare
narratio session validate 2026-04-04
```
Safe Fix:
- configure exactly one audio source mode.
- verify `.flac` files and storage access.
Safe fix:
Links:
- verify prefix contents and storage access;
- keep session audio mode consistent.
Relevant reference: [Operations](./operations.md).
## References
- [docs/cli.md](./cli.md)
- [docs/config.md](./config.md)
- [docs/operations.md](./operations.md)
## Archive promotion/current-pointer failure
Symptom:
- archive fails on required promotion source missing or pointer write failure.
Likely Cause:
- required promoted file absent (including analyze outputs not generated for this run).
- storage upload failed before `current/run_id.txt` commit marker write.
Diagnostics:
```bash
narratio status --manifest /path/to/manifest.json
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 archive
```
Safe Fix:
- rerun or resume upstream stages to generate required files.
- adjust promotion `source`/`dest` rules to match artifacts that must exist.
- retry after storage issue is resolved.
Links:
- [docs/operations.md](./operations.md)
- [docs/config.md](./config.md)
- [docs/internal/stage-archive.md](./internal/stage-archive.md)
- [docs/internal/stage-publish.md](./internal/stage-publish.md)

48
examples/README.md Normal file
View File

@@ -0,0 +1,48 @@
# Maintained Examples
These files are safe, copyable starting points for Narratio configuration and
input structure. Replace placeholder identifiers, storage names, integration
URLs, and paths for the target environment. Field meanings and defaults belong
in the [configuration reference](../docs/config.md).
## Pipeline Configuration
- [Minimal pipeline](pipeline.minimal.yml): campaign discovery plus the required
WhisperX URL.
- [Production-shaped pipeline](pipeline.production.yml): S3 storage, publish,
external tools, and configured Scriptorium artifacts.
- [Full annotated pipeline](pipeline.full.annotated.yml): every implemented
pipeline section with explanatory comments.
- [Extraction subset pipeline](pipeline.extraction-subset.yml): a focused
Scriptorium artifact consuming only three declared Notarius lanes.
The existing `internal/config` example test loads and validates each pipeline
with the sample campaign and a compatible local- or S3-audio session.
## Campaign And Session Configuration
- [Sample campaign](campaigns/sample-campaign/campaign.yml), its
[session template](campaigns/sample-campaign/session.template.yml), and its
adjacent stable inputs provide a complete campaign directory shape.
- [Local-audio session](session.local-audio.yml) and
[S3-audio session](session.s3-audio.yml) are concrete session files.
- [Session template](session.template.yml) and the campaign-local equivalent
demonstrate the narrow placeholder syntax consumed by `session init`; they
are templates, not runtime session files.
## Input Fixtures
- [Speakers](speakers.yml), [autocorrect](autocorrect.yml), and
[glossary](glossary.yml) show the standalone input shapes.
- The sample campaign references its local
[speakers](campaigns/sample-campaign/speakers.yml),
[autocorrect](campaigns/sample-campaign/autocorrect.yml),
[glossary](campaigns/sample-campaign/glossary.yml),
[players](campaigns/sample-campaign/players.yml), and
[party](campaigns/sample-campaign/party.yml) fixtures.
- [Sample speaker audio](audio/sample-speaker.flac) is a text placeholder that
reserves the expected filename and directory shape. Replace it with a real
FLAC file before running transcription.
The examples contain environment-variable names but no credential values. They
use fictional campaign content and reserved example domains.

View File

@@ -0,0 +1 @@
[]

View File

@@ -0,0 +1,8 @@
campaign_id: sample-campaign
session_template_file: ./session.template.yml
inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
players_file: ./players.yml
party_file: ./party.yml

View File

@@ -0,0 +1 @@
[]

View File

@@ -0,0 +1,2 @@
- name: Example Hero
type: pc

View File

@@ -0,0 +1,2 @@
- name: Example Player
role: player

View File

@@ -0,0 +1,3 @@
session_id: "{{ session_id }}"
inputs:
audio_dir: ./audio

View File

@@ -0,0 +1,5 @@
match:
- speaker: "Example Speaker"
match:
- "Example_Speaker"
- "Example"

View File

@@ -0,0 +1,55 @@
# Purpose-specific extraction example: a Scriptorium session brief consumes
# only the three Notarius lanes it needs.
campaigns:
root: /usr/local/share/narratio/campaigns
default_campaign_id: sample-campaign
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"
notarius:
enabled: true
binary: notarius
config_path: /usr/local/etc/notarius/config.yml
pipeline_id: dnd-session
timeout: 3h
outputs:
npc_registry:
lane_id: npc-registry
media_type: application/json
schema_id: notarius.dnd.npc_registry
schema_version: v1
module_key: dnd/npc-registry
location_registry:
lane_id: location-registry
media_type: application/json
schema_id: notarius.dnd.location_registry
schema_version: v1
module_key: dnd/location-registry
scene_descriptions:
lane_id: scene-descriptions
media_type: application/json
schema_id: notarius.dnd.scene_descriptions
schema_version: v1
module_key: dnd/scene-descriptions
scriptorium:
binary: scriptorium
config_path: /usr/local/etc/scriptorium/config.yml
artifacts:
session_brief:
enabled: true
prompt_id: dnd.session_brief
output_path: artifacts/session_brief.md
inputs:
npcs:
source: narratio.extraction.npc_registry
required: true
locations:
source: narratio.extraction.location_registry
required: true
scenes:
source: narratio.extraction.scene_descriptions
required: true

View File

@@ -4,21 +4,18 @@
workspace:
# Optional: defaults to /var/lib/narratio.
root: /var/lib/narratio/workspace
# Optional: remove run-scoped workdir after successful archive commit.
cleanup_after_archive: false
# Optional: remove run-scoped workdir after successful publish commit.
cleanup_after_publish: false
# Optional: local secret file loader (directory of ENV_VAR_NAME files).
# secrets:
# env_dir: ./secrets
storage:
# Optional storage backend selector; use "s3" for archive + S3 audio workflows.
# Optional storage backend selector; use "s3" for publish + S3 audio workflows.
backend: s3
# Compatibility fields retained in schema.
bucket: ""
prefix: ""
s3:
# Required when using S3 audio or S3 archive uploads.
# Required when using S3 audio or S3 publish uploads.
bucket: my-dnd-archive
# Optional; defaults to "dnd".
root_prefix: dnd
@@ -30,20 +27,32 @@ storage:
access_key_id_env: OBJECT_STORAGE_KEY_ID
secret_access_key_env: OBJECT_STORAGE_KEY
campaigns:
# Optional; defaults to /usr/local/share/narratio/campaigns.
root: /usr/local/share/narratio/campaigns
# Optional command default when --campaign is omitted.
default_campaign_id: sample-campaign
spool:
# Optional; defaults to /var/spool/narratio.
root: /var/spool/narratio
# Optional cleanup of run-scoped spool audio after successful archive commit.
delete_audio_after_archive: false
# Optional cleanup of run-scoped spool audio after successful publish commit.
delete_audio_after_publish: false
archive:
publish:
# Optional booleans; defaults are true.
enabled: true
upload_run: true
# Optional promotion rules; sources use Narratio artifact source IDs.
promote_artifacts:
- source: narratio.transcript.trimmed
dest: transcripts/trimmed.json
# Optional publish output rules; sources use Narratio artifact source IDs.
outputs:
- source: narratio.transcript.final_trimmed
dest: transcripts/final.trimmed.json
required: true
- source: narratio.transcript.final_markdown
dest: transcripts/final.md
required: true
- source: narratio.transcript.final_trimmed_markdown
dest: transcripts/final.trimmed.md
required: true
- source: narratio.artifact.session_recap
dest: artifacts/session_recap.md
@@ -51,6 +60,11 @@ archive:
- source: narratio.artifact.player_handout
dest: artifacts/player_handout.md
required: false
# Extraction lanes publish only when named explicitly; the bundle and index
# are never implicit publish sources.
- source: narratio.extraction.npc_registry
dest: artifacts/extraction/npc-registry.json
required: true
whisperx:
# Required.
@@ -96,25 +110,96 @@ audita:
normalize:
# Optional; defaults shown explicitly.
output_path: transcripts/normalized.json
output_path: transcripts/final.json
output_schema: seriatim-intermediate
report: true
trim:
# Keep disabled unless bounds prompt integration is configured.
enabled: false
output_path: transcripts/trimmed.json
# Optional; defaults shown explicitly.
enabled: true
output_path: transcripts/final.trimmed.json
bounds:
prompt_id: dnd.session_bounds
profile_id: local-fast
profile_id: ""
transcript_input_name: transcript
output_path: reports/session_bounds.json
output_path: artifacts/session_bounds.json
timeout: 10m
render_debug: false
render_output_path: reports/session_bounds.render.json
seriatim:
report: false
notarius:
# Optional structured extraction between trim and render.
enabled: true
binary: notarius
config_path: /usr/local/etc/notarius/config.yml
pipeline_id: dnd-session
timeout: 3h
working_directory: /usr/local/etc/notarius
# Each key creates source narratio.extraction.<key>. These constraints match
# the current Notarius D&D lane contracts; update them with Notarius.
outputs:
item_registry:
lane_id: item-registry
media_type: application/json
schema_id: notarius.dnd.item_registry
schema_version: v1
module_key: dnd/item-registry
npc_registry:
lane_id: npc-registry
media_type: application/json
schema_id: notarius.dnd.npc_registry
schema_version: v1
module_key: dnd/npc-registry
location_registry:
lane_id: location-registry
media_type: application/json
schema_id: notarius.dnd.location_registry
schema_version: v1
module_key: dnd/location-registry
scene_descriptions:
lane_id: scene-descriptions
media_type: application/json
schema_id: notarius.dnd.scene_descriptions
schema_version: v1
module_key: dnd/scene-descriptions
item_occurrences:
lane_id: item-occurrences
media_type: application/json
schema_id: notarius.dnd.item_occurrences
schema_version: v1
module_key: dnd/item-occurrences
spells:
lane_id: spells
media_type: application/json
schema_id: notarius.dnd.spells
schema_version: v1
module_key: dnd/spells
combat_turns:
lane_id: combat-turns
media_type: application/json
schema_id: notarius.dnd.combat_turns
schema_version: v1
module_key: dnd/combat-turns
npc_occurrences:
lane_id: npc-occurrences
media_type: application/json
schema_id: notarius.dnd.npc_occurrences
schema_version: v1
module_key: dnd/npc-occurrences
location_occurrences:
lane_id: location-occurrences
media_type: application/json
schema_id: notarius.dnd.location_occurrences
schema_version: v1
module_key: dnd/location-occurrences
enemy_events:
lane_id: enemy-events
media_type: application/json
schema_id: notarius.dnd.enemy_events
schema_version: v1
module_key: dnd/enemy-events
scriptorium:
binary: scriptorium
config_path: /usr/local/etc/scriptorium/config.yml
@@ -130,12 +215,19 @@ scriptorium:
timeout: 10m
inputs:
transcript:
source: narratio.transcript.trimmed
source: narratio.transcript.final_trimmed
required: true
previous_recap:
source: previous_session_artifact
artifact: session_recap
path: ""
source: narratio.previous_session.artifact.session_recap
required: false
players:
source: narratio.input.players
required: true
party:
source: narratio.input.party
required: true
glossary:
source: narratio.input.glossary
required: false
vars:
session_id: true
@@ -160,21 +252,13 @@ scriptorium:
source: narratio.artifact.session_recap
required: true
transcript:
source: narratio.transcript.trimmed
source: narratio.transcript.final_trimmed
required: true
vars:
session_id: true
campaign_name: true
output_kind: player_handout
analyzer:
# Optional adapter settings.
binary_path: ""
timeout: 2m
artifacts:
output_dir: ""
types: []
notification:
# Optional notification settings.
backend: ""

View File

@@ -1,2 +1,6 @@
campaigns:
root: /usr/local/share/narratio/campaigns
default_campaign_id: sample-campaign
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"

View File

@@ -1,6 +1,6 @@
workspace:
root: /var/lib/narratio/workspace
cleanup_after_archive: true
cleanup_after_publish: true
storage:
backend: s3
@@ -11,16 +11,26 @@ storage:
access_key_id_env: OBJECT_STORAGE_KEY_ID
secret_access_key_env: OBJECT_STORAGE_KEY
campaigns:
root: /usr/local/share/narratio/campaigns
default_campaign_id: sample-campaign
spool:
root: /var/spool/narratio
delete_audio_after_archive: true
delete_audio_after_publish: true
archive:
publish:
enabled: true
upload_run: true
promote_artifacts:
- source: narratio.transcript.trimmed
dest: transcripts/trimmed.json
outputs:
- source: narratio.transcript.final_trimmed
dest: transcripts/final.trimmed.json
required: true
- source: narratio.transcript.final_markdown
dest: transcripts/final.md
required: true
- source: narratio.transcript.final_trimmed_markdown
dest: transcripts/final.trimmed.md
required: true
- source: narratio.artifact.session_recap
dest: artifacts/session_recap.md
@@ -57,13 +67,10 @@ audita:
report: true
normalize:
output_path: transcripts/normalized.json
output_path: transcripts/final.json
output_schema: seriatim-intermediate
report: true
trim:
enabled: false
scriptorium:
binary: scriptorium
config_path: /usr/local/etc/scriptorium/config.yml
@@ -78,11 +85,19 @@ scriptorium:
timeout: 10m
inputs:
transcript:
source: narratio.transcript.trimmed
source: narratio.transcript.final_trimmed
required: true
previous_recap:
source: previous_session_artifact
artifact: session_recap
source: narratio.previous_session.artifact.session_recap
required: false
players:
source: narratio.input.players
required: true
party:
source: narratio.input.party
required: true
glossary:
source: narratio.input.glossary
required: false
vars:
session_id: true
@@ -103,14 +118,11 @@ scriptorium:
source: narratio.artifact.session_recap
required: true
transcript:
source: narratio.transcript.trimmed
source: narratio.transcript.final_trimmed
required: true
vars:
session_id: true
output_kind: player_handout
analyzer:
timeout: 2m
notification:
timeout: 30s

View File

@@ -1,9 +1,5 @@
session_id: 2026-05-03
campaign: sample-campaign
date: 2026-05-03
title: Sample Session
inputs:
audio_dir: ./audio
speakers_file: ./examples/speakers.yml
autocorrect_file: ./examples/autocorrect.yml
glossary_file: ./examples/glossary.yml

View File

@@ -1,10 +1,6 @@
session_id: 2026-05-03
campaign: sample-campaign
date: 2026-05-03
title: Sample Session
inputs:
audio_s3:
prefix: audio/
speakers_file: ./examples/speakers.yml
autocorrect_file: ./examples/autocorrect.yml
glossary_file: ./examples/glossary.yml

View File

@@ -1,7 +1,3 @@
session_id: "{{ session_id }}"
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./examples/speakers.yml
autocorrect_file: ./examples/autocorrect.yml
glossary_file: ./examples/glossary.yml

View File

@@ -1,5 +1,5 @@
match:
- speaker: "Eric Rakestraw"
- speaker: "Example Speaker"
match:
- "Eric_Rakestraw"
- "Eric"
- "Example_Speaker"
- "Example"

1
go.mod
View File

@@ -7,6 +7,7 @@ require (
github.com/aws/aws-sdk-go-v2/credentials v1.19.16
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0
github.com/aws/smithy-go v1.25.1
golang.org/x/sys v0.47.0
gopkg.in/yaml.v3 v3.0.1
)

2
go.sum
View File

@@ -34,6 +34,8 @@ github.com/aws/aws-sdk-go-v2/service/sts v1.42.1 h1:F/M5Y9I3nwr2IEpshZgh1GeHpOIt
github.com/aws/aws-sdk-go-v2/service/sts v1.42.1/go.mod h1:mTNxImtovCOEEuD65mKW7DCsL+2gjEH+RPEAexAzAio=
github.com/aws/smithy-go v1.25.1 h1:J8ERsGSU7d+aCmdQur5Txg6bVoYelvQJgtZehD12GkI=
github.com/aws/smithy-go v1.25.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc=
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=

View File

@@ -1,40 +0,0 @@
package analyzer
import "context"
// NoopRunner is a deterministic no-op analyzer adapter.
type NoopRunner struct{}
// Run returns the requested output path with placeholder metadata.
func (n *NoopRunner) Run(ctx context.Context, req AnalyzeRequest) (AnalyzeResult, error) {
if err := ctx.Err(); err != nil {
return AnalyzeResult{}, err
}
return AnalyzeResult{ArtifactPath: req.OutputPath, Metadata: map[string]any{"placeholder": true}}, nil
}
// FakeRunner captures analyze requests and returns deterministic responses.
type FakeRunner struct {
Requests []AnalyzeRequest
Err error
Result AnalyzeResult
}
// Run records request and returns configured response.
func (f *FakeRunner) Run(ctx context.Context, req AnalyzeRequest) (AnalyzeResult, error) {
if err := ctx.Err(); err != nil {
return AnalyzeResult{}, err
}
f.Requests = append(f.Requests, req)
if f.Err != nil {
return AnalyzeResult{}, f.Err
}
res := f.Result
if res.ArtifactPath == "" {
res.ArtifactPath = req.OutputPath
}
if res.Metadata == nil {
res.Metadata = map[string]any{"fake": true}
}
return res, nil
}

View File

@@ -1,31 +0,0 @@
package analyzer
import (
"context"
"errors"
"testing"
)
func TestFakeRunnerCapturesRequestAndReturnsPath(t *testing.T) {
fake := &FakeRunner{}
req := AnalyzeRequest{ArtifactType: "session-log", OutputPath: "artifacts/session-log.md"}
res, err := fake.Run(context.Background(), req)
if err != nil {
t.Fatalf("Run() error = %v", err)
}
if len(fake.Requests) != 1 || fake.Requests[0].ArtifactType != "session-log" {
t.Fatalf("requests = %#v, want captured request", fake.Requests)
}
if res.ArtifactPath != req.OutputPath {
t.Fatalf("artifact path = %q, want %q", res.ArtifactPath, req.OutputPath)
}
}
func TestFakeRunnerError(t *testing.T) {
fake := &FakeRunner{Err: errors.New("boom")}
_, err := fake.Run(context.Background(), AnalyzeRequest{})
if err == nil {
t.Fatal("expected error, got nil")
}
}

View File

@@ -1,28 +0,0 @@
// Package analyzer declares the adapter contract for artifact analysis generation.
package analyzer
import "context"
// TODO: implement analyzer integration once the analyzer contract is finalized.
// Runner is the adapter boundary for analyzer invocations.
type Runner interface {
Run(ctx context.Context, req AnalyzeRequest) (AnalyzeResult, error)
}
// AnalyzeRequest describes one analyzer artifact generation request.
type AnalyzeRequest struct {
ArtifactType string
ProcessedTranscriptPath string
ContextReferences []string
OutputPath string
GeneratedConfigPath string
StdoutLogPath string
StderrLogPath string
}
// AnalyzeResult describes analyzer output.
type AnalyzeResult struct {
ArtifactPath string
Metadata map[string]any
}

View File

@@ -14,7 +14,7 @@ func TestFakeRunnerCapturesRequestAndReturnsPath(t *testing.T) {
dir := t.TempDir()
req := PolishRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "audita.yml"),
OutputProcessedPath: filepath.Join(dir, "transcripts", "processed.json"),
OutputProcessedPath: filepath.Join(dir, "transcripts", "polished.json"),
StdoutLogPath: filepath.Join(dir, "logs", "audita.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "audita.stderr.log"),
}

View File

@@ -52,9 +52,9 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
dir := t.TempDir()
req := PolishRequest{
GeneratedConfigPath: filepath.Join(dir, "audita.generated.yml"),
MergedTranscriptPath: filepath.Join(dir, "merged.json"),
MergedTranscriptPath: filepath.Join(dir, "base.json"),
GlossaryPath: filepath.Join(dir, "glossary.yml"),
OutputProcessedPath: filepath.Join(dir, "processed.json"),
OutputProcessedPath: filepath.Join(dir, "polished.json"),
ReportPath: filepath.Join(dir, "audita.report.json"),
WorkDir: filepath.Join(dir, "artifacts", "audita-work"),
StdoutLogPath: filepath.Join(dir, "audita.stdout.log"),
@@ -571,7 +571,7 @@ func mustAuditaRunner(t *testing.T, cfg SubprocessRunnerConfig) *SubprocessRunne
func auditaReqForTest(t *testing.T, withReport bool) PolishRequest {
t.Helper()
dir := t.TempDir()
merged := filepath.Join(dir, "merged.json")
merged := filepath.Join(dir, "base.json")
glossary := filepath.Join(dir, "glossary.yml")
writeAuditaTestFile(t, merged, `{"segments":[]}`)
writeAuditaTestFile(t, glossary, "terms: []\n")
@@ -579,7 +579,7 @@ func auditaReqForTest(t *testing.T, withReport bool) PolishRequest {
GeneratedConfigPath: filepath.Join(dir, "audita.generated.yml"),
MergedTranscriptPath: merged,
GlossaryPath: glossary,
OutputProcessedPath: filepath.Join(dir, "processed.json"),
OutputProcessedPath: filepath.Join(dir, "polished.json"),
WorkDir: filepath.Join(dir, "artifacts", "audita-work"),
StdoutLogPath: filepath.Join(dir, "audita.stdout.log"),
StderrLogPath: filepath.Join(dir, "audita.stderr.log"),

View File

@@ -0,0 +1,22 @@
package notarius
import "context"
// FakeRunner is a configurable in-memory runner for stage tests.
type FakeRunner struct {
Requests []RunRequest
Result RunResult
Err error
}
// Run records the request and returns the configured result or error.
func (f *FakeRunner) Run(ctx context.Context, req RunRequest) (RunResult, error) {
if err := ctx.Err(); err != nil {
return RunResult{}, err
}
f.Requests = append(f.Requests, req)
if f.Err != nil {
return RunResult{}, f.Err
}
return f.Result, nil
}

View File

@@ -0,0 +1,108 @@
// Package notarius declares the adapter contract for Notarius CLI invocations.
package notarius
import (
"context"
"time"
)
const ReceiptSchemaVersion = "notarius.run-result.v1"
// Runner is the adapter boundary for a complete Notarius pipeline invocation.
type Runner interface {
Run(ctx context.Context, req RunRequest) (RunResult, error)
}
// RunRequest contains the resolved inputs and diagnostic destinations for one invocation.
type RunRequest struct {
Binary string
ConfigPath string
PipelineID string
InputPath string
OutputRoot string
WorkingDirectory string
ReceiptPath string
LogPath string
Timeout time.Duration
}
// Receipt is the transport-neutral successful run receipt.
type Receipt struct {
SchemaVersion string
RunID string
PipelineID string
OutputDirectory string
IndexFile string
NormalizedOutputCount int
RejectedOutputCount int
WarningCount int
ValidationStatus string
DebugDirectory string
}
// LaneDescriptor identifies one normalized lane payload discovered through the index.
type LaneDescriptor struct {
LaneID string
File string
Path string
MediaType string
ModuleKey string
SchemaID string
SchemaName string
SchemaVersion string
}
// PipelineDescriptor identifies a pipeline-wide artifact discovered through the index.
type PipelineDescriptor struct {
ArtifactKind string
File string
Path string
MediaType string
SchemaID string
SchemaName string
SchemaVersion string
}
// Index describes the validated bundle-management and artifact paths.
type Index struct {
Path string
ManifestFile string
ManifestPath string
RejectedFile string
RejectedPath string
WarningsFile string
WarningsPath string
Lanes []LaneDescriptor
ChunkMap *PipelineDescriptor
EvidenceContext *PipelineDescriptor
}
// RejectionSummary retains structured rejection identity without free-form messages.
type RejectionSummary struct {
Stage string
StepID string
LaneID string
ModuleKey string
ChunkID string
ValidatorName string
ReasonCode string
}
// WarningSummary retains structured warning identity without free-form messages.
type WarningSummary struct {
Scope string
ReasonCode string
}
// RunResult describes a successfully decoded and validated Notarius bundle.
type RunResult struct {
Receipt Receipt
Index Index
BundleRoot string
ReceiptPath string
LogPath string
ExitCode int
Duration time.Duration
Rejections []RejectionSummary
Warnings []WarningSummary
}

View File

@@ -0,0 +1,524 @@
package notarius
import (
"context"
"encoding/json"
"errors"
"fmt"
"io"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
"gitea.maximumdirect.net/eric/narratio/internal/pathsafe"
)
const (
maxReceiptBytes = 1 << 20
maxIndexBytes = 4 << 20
maxSummaryBytes = 4 << 20
canonicalIndexFile = "index.json"
canonicalManifestFile = "manifest.json"
canonicalRejectedFile = "rejected.json"
canonicalWarningsFile = "warnings.json"
)
type subprocessRun func(context.Context, subprocess.RunRequest) (subprocess.RunResult, error)
// SubprocessRunner invokes Notarius through its public CLI.
type SubprocessRunner struct {
run subprocessRun
}
// NewSubprocessRunner constructs a production Notarius subprocess runner.
func NewSubprocessRunner() *SubprocessRunner {
return &SubprocessRunner{run: subprocess.Run}
}
// Run executes a complete Notarius pipeline and discovers its published bundle.
func (r *SubprocessRunner) Run(ctx context.Context, req RunRequest) (RunResult, error) {
if r == nil || r.run == nil {
return RunResult{}, fmt.Errorf("notarius subprocess runner is nil")
}
if err := validateRunRequest(req); err != nil {
return RunResult{}, err
}
args := []string{
"run", req.PipelineID,
"--config", req.ConfigPath,
"--input", req.InputPath,
"--output-dir", req.OutputRoot,
"--json",
}
processResult, err := r.run(ctx, subprocess.RunRequest{
Executable: req.Binary,
Args: args,
WorkingDir: req.WorkingDirectory,
Timeout: req.Timeout,
StdoutLogPath: req.ReceiptPath,
StderrLogPath: req.LogPath,
})
baseResult := RunResult{
ReceiptPath: req.ReceiptPath,
LogPath: req.LogPath,
ExitCode: processResult.ExitCode,
Duration: processResult.Duration,
}
if err != nil {
return baseResult, fmt.Errorf("run notarius pipeline %q: %w", req.PipelineID, err)
}
receipt, err := loadReceipt(req.ReceiptPath, req.PipelineID)
if err != nil {
return baseResult, err
}
bundleRoot, err := validateBundleRoot(req.OutputRoot, receipt.OutputDirectory)
if err != nil {
return baseResult, err
}
indexPath, err := resolveRegularFile(bundleRoot, receipt.IndexFile)
if err != nil {
return baseResult, fmt.Errorf("resolve receipt index file: %w", err)
}
index, err := loadIndex(bundleRoot, indexPath)
if err != nil {
return baseResult, err
}
rejections, err := loadRejections(index.RejectedPath)
if err != nil {
return baseResult, err
}
warnings, err := loadWarnings(index.WarningsPath)
if err != nil {
return baseResult, err
}
baseResult.Receipt = receipt
baseResult.Index = index
baseResult.BundleRoot = bundleRoot
baseResult.Rejections = rejections
baseResult.Warnings = warnings
return baseResult, nil
}
func validateRunRequest(req RunRequest) error {
if strings.TrimSpace(req.Binary) == "" {
return fmt.Errorf("notarius binary is required")
}
if strings.TrimSpace(req.PipelineID) == "" {
return fmt.Errorf("notarius pipeline id is required")
}
if req.Timeout <= 0 {
return fmt.Errorf("notarius timeout must be positive")
}
for label, path := range map[string]string{
"config": req.ConfigPath,
"input": req.InputPath,
"output root": req.OutputRoot,
"working directory": req.WorkingDirectory,
"receipt": req.ReceiptPath,
"log": req.LogPath,
} {
if strings.TrimSpace(path) == "" {
return fmt.Errorf("notarius %s path is required", label)
}
if !filepath.IsAbs(path) {
return fmt.Errorf("notarius %s path must be absolute", label)
}
}
if filepath.Clean(req.ReceiptPath) == filepath.Clean(req.LogPath) {
return fmt.Errorf("notarius receipt and log paths must be different")
}
if err := requireRegularFile(req.ConfigPath); err != nil {
return fmt.Errorf("validate notarius config path: %w", err)
}
if err := requireRegularFile(req.InputPath); err != nil {
return fmt.Errorf("validate notarius input path: %w", err)
}
if err := requireDirectory(req.OutputRoot); err != nil {
return fmt.Errorf("validate notarius output root: %w", err)
}
if err := requireDirectory(req.WorkingDirectory); err != nil {
return fmt.Errorf("validate notarius working directory: %w", err)
}
if err := validateLogDestination(req.ReceiptPath); err != nil {
return fmt.Errorf("validate notarius receipt path: %w", err)
}
if err := validateLogDestination(req.LogPath); err != nil {
return fmt.Errorf("validate notarius log path: %w", err)
}
return nil
}
type receiptDocument struct {
SchemaVersion string `json:"schema_version"`
RunID string `json:"run_id"`
PipelineID string `json:"pipeline_id"`
OutputDirectory string `json:"output_directory"`
IndexFile string `json:"index_file"`
NormalizedOutputCount *int `json:"normalized_output_count"`
RejectedOutputCount *int `json:"rejected_output_count"`
WarningCount *int `json:"warning_count"`
ValidationStatus string `json:"validation_status"`
DebugDirectory string `json:"debug_directory"`
}
func loadReceipt(path, pipelineID string) (Receipt, error) {
var document receiptDocument
if err := decodeBoundedJSON(path, maxReceiptBytes, &document); err != nil {
return Receipt{}, fmt.Errorf("decode notarius receipt: %w", err)
}
if document.SchemaVersion != ReceiptSchemaVersion {
return Receipt{}, fmt.Errorf("unsupported notarius receipt schema version %q", document.SchemaVersion)
}
if strings.TrimSpace(document.RunID) == "" || strings.TrimSpace(document.PipelineID) == "" ||
strings.TrimSpace(document.OutputDirectory) == "" || strings.TrimSpace(document.ValidationStatus) == "" ||
document.NormalizedOutputCount == nil ||
document.RejectedOutputCount == nil || document.WarningCount == nil {
return Receipt{}, fmt.Errorf("notarius receipt is missing required fields")
}
if document.IndexFile != canonicalIndexFile {
return Receipt{}, fmt.Errorf("notarius receipt index_file %q is incompatible; want %q", document.IndexFile, canonicalIndexFile)
}
if document.PipelineID != pipelineID {
return Receipt{}, fmt.Errorf("notarius receipt pipeline id %q does not match requested pipeline %q", document.PipelineID, pipelineID)
}
if *document.NormalizedOutputCount < 0 || *document.RejectedOutputCount < 0 || *document.WarningCount < 0 {
return Receipt{}, fmt.Errorf("notarius receipt counts must be non-negative")
}
if !filepath.IsAbs(document.OutputDirectory) {
return Receipt{}, fmt.Errorf("notarius receipt output directory must be absolute")
}
if document.DebugDirectory != "" && !filepath.IsAbs(document.DebugDirectory) {
return Receipt{}, fmt.Errorf("notarius receipt debug directory must be absolute when present")
}
return Receipt{
SchemaVersion: document.SchemaVersion,
RunID: document.RunID,
PipelineID: document.PipelineID,
OutputDirectory: filepath.Clean(document.OutputDirectory),
IndexFile: document.IndexFile,
NormalizedOutputCount: *document.NormalizedOutputCount,
RejectedOutputCount: *document.RejectedOutputCount,
WarningCount: *document.WarningCount,
ValidationStatus: document.ValidationStatus,
DebugDirectory: document.DebugDirectory,
}, nil
}
type indexDocument struct {
ManifestFile string `json:"manifest_file"`
OutputFiles *[]laneDocument `json:"output_files"`
RejectedFile string `json:"rejected_file"`
WarningsFile string `json:"warnings_file"`
ChunkMap *pipelineDocument `json:"chunk_map"`
EvidenceContext *pipelineDocument `json:"evidence_context"`
}
type laneDocument struct {
LaneID string `json:"lane_id"`
File string `json:"file"`
MediaType string `json:"media_type"`
ModuleKey string `json:"module_key"`
SchemaID string `json:"schema_id"`
SchemaName string `json:"schema_name"`
SchemaVersion string `json:"schema_version"`
}
type pipelineDocument struct {
ArtifactKind string `json:"artifact_kind"`
File string `json:"file"`
MediaType string `json:"media_type"`
SchemaID string `json:"schema_id"`
SchemaName string `json:"schema_name"`
SchemaVersion string `json:"schema_version"`
}
func loadIndex(bundleRoot, indexPath string) (Index, error) {
var document indexDocument
if err := decodeBoundedJSON(indexPath, maxIndexBytes, &document); err != nil {
return Index{}, fmt.Errorf("decode notarius index: %w", err)
}
for _, field := range []struct {
name string
got string
want string
}{
{name: "manifest_file", got: document.ManifestFile, want: canonicalManifestFile},
{name: "rejected_file", got: document.RejectedFile, want: canonicalRejectedFile},
{name: "warnings_file", got: document.WarningsFile, want: canonicalWarningsFile},
} {
if field.got != field.want {
return Index{}, fmt.Errorf("notarius index %s %q is incompatible; want %q", field.name, field.got, field.want)
}
}
if document.OutputFiles == nil {
return Index{}, fmt.Errorf("notarius index is missing required output_files")
}
index := Index{
Path: indexPath,
ManifestFile: document.ManifestFile,
RejectedFile: document.RejectedFile,
WarningsFile: document.WarningsFile,
}
var err error
if index.ManifestPath, err = resolveRegularFile(bundleRoot, index.ManifestFile); err != nil {
return Index{}, fmt.Errorf("resolve notarius manifest file: %w", err)
}
if index.RejectedPath, err = resolveRegularFile(bundleRoot, index.RejectedFile); err != nil {
return Index{}, fmt.Errorf("resolve notarius rejection file: %w", err)
}
if index.WarningsPath, err = resolveRegularFile(bundleRoot, index.WarningsFile); err != nil {
return Index{}, fmt.Errorf("resolve notarius warning file: %w", err)
}
seenLanes := make(map[string]struct{}, len(*document.OutputFiles))
for _, lane := range *document.OutputFiles {
if strings.TrimSpace(lane.LaneID) == "" || strings.TrimSpace(lane.File) == "" {
return Index{}, fmt.Errorf("notarius lane descriptors require lane_id and file")
}
if _, exists := seenLanes[lane.LaneID]; exists {
return Index{}, fmt.Errorf("notarius index contains duplicate lane id %q", lane.LaneID)
}
seenLanes[lane.LaneID] = struct{}{}
path, err := resolveRegularFile(bundleRoot, lane.File)
if err != nil {
return Index{}, fmt.Errorf("resolve notarius lane %q file: %w", lane.LaneID, err)
}
index.Lanes = append(index.Lanes, LaneDescriptor{
LaneID: lane.LaneID, File: lane.File, Path: path, MediaType: lane.MediaType,
ModuleKey: lane.ModuleKey, SchemaID: lane.SchemaID, SchemaName: lane.SchemaName,
SchemaVersion: lane.SchemaVersion,
})
}
if document.ChunkMap != nil {
index.ChunkMap, err = resolvePipelineDescriptor(bundleRoot, "chunk_map", *document.ChunkMap)
if err != nil {
return Index{}, err
}
}
if document.EvidenceContext != nil {
index.EvidenceContext, err = resolvePipelineDescriptor(bundleRoot, "evidence_context", *document.EvidenceContext)
if err != nil {
return Index{}, err
}
}
return index, nil
}
func resolvePipelineDescriptor(bundleRoot, label string, document pipelineDocument) (*PipelineDescriptor, error) {
if strings.TrimSpace(document.ArtifactKind) == "" || strings.TrimSpace(document.File) == "" ||
strings.TrimSpace(document.MediaType) == "" || strings.TrimSpace(document.SchemaID) == "" ||
strings.TrimSpace(document.SchemaName) == "" || strings.TrimSpace(document.SchemaVersion) == "" {
return nil, fmt.Errorf("notarius %s descriptor is missing required fields", label)
}
path, err := resolveRegularFile(bundleRoot, document.File)
if err != nil {
return nil, fmt.Errorf("resolve notarius %s file: %w", label, err)
}
return &PipelineDescriptor{
ArtifactKind: document.ArtifactKind, File: document.File, Path: path,
MediaType: document.MediaType, SchemaID: document.SchemaID,
SchemaName: document.SchemaName, SchemaVersion: document.SchemaVersion,
}, nil
}
type rejectionDocument struct {
Rejected *[]struct {
Stage string `json:"stage"`
StepID string `json:"step_id"`
LaneID string `json:"lane_id"`
ModuleKey string `json:"module_key"`
ChunkID string `json:"chunk_id"`
ValidatorName string `json:"validator_name"`
ReasonCode string `json:"reason_code"`
Message string `json:"message"`
} `json:"rejected"`
}
func loadRejections(path string) ([]RejectionSummary, error) {
var document rejectionDocument
if err := decodeBoundedJSON(path, maxSummaryBytes, &document); err != nil {
return nil, fmt.Errorf("decode notarius rejections: %w", err)
}
if document.Rejected == nil {
return nil, fmt.Errorf("notarius rejection document is missing rejected array")
}
summaries := make([]RejectionSummary, 0, len(*document.Rejected))
for _, item := range *document.Rejected {
if strings.TrimSpace(item.Stage) == "" || strings.TrimSpace(item.Message) == "" {
return nil, fmt.Errorf("notarius rejection entries require stage and message")
}
summaries = append(summaries, RejectionSummary{
Stage: item.Stage, StepID: item.StepID, LaneID: item.LaneID,
ModuleKey: item.ModuleKey, ChunkID: item.ChunkID,
ValidatorName: item.ValidatorName, ReasonCode: item.ReasonCode,
})
}
return summaries, nil
}
type warningDocument struct {
Warnings *[]struct {
Scope string `json:"scope"`
ReasonCode string `json:"reason_code"`
Message string `json:"message"`
} `json:"warnings"`
}
func loadWarnings(path string) ([]WarningSummary, error) {
var document warningDocument
if err := decodeBoundedJSON(path, maxSummaryBytes, &document); err != nil {
return nil, fmt.Errorf("decode notarius warnings: %w", err)
}
if document.Warnings == nil {
return nil, fmt.Errorf("notarius warning document is missing warnings array")
}
summaries := make([]WarningSummary, 0, len(*document.Warnings))
for _, item := range *document.Warnings {
if strings.TrimSpace(item.ReasonCode) == "" || strings.TrimSpace(item.Message) == "" {
return nil, fmt.Errorf("notarius warning entries require reason_code and message")
}
summaries = append(summaries, WarningSummary{Scope: item.Scope, ReasonCode: item.ReasonCode})
}
return summaries, nil
}
func decodeBoundedJSON(path string, limit int64, destination any) error {
inspected, err := os.Lstat(path)
if err != nil {
return err
}
if inspected.Mode()&os.ModeSymlink != 0 || !inspected.Mode().IsRegular() {
return fmt.Errorf("path %q must be a regular file without symlinks", path)
}
file, err := os.Open(path)
if err != nil {
return err
}
defer func() { _ = file.Close() }()
opened, err := file.Stat()
if err != nil {
return err
}
if !opened.Mode().IsRegular() || !os.SameFile(inspected, opened) {
return fmt.Errorf("file %q changed before it could be read", path)
}
reader := io.LimitReader(file, limit+1)
data, err := io.ReadAll(reader)
if err != nil {
return err
}
if int64(len(data)) > limit {
return fmt.Errorf("file %q exceeds %d-byte limit", path, limit)
}
if err := json.Unmarshal(data, destination); err != nil {
return err
}
return nil
}
func validateBundleRoot(outputRoot, bundleRoot string) (string, error) {
root := filepath.Clean(outputRoot)
bundle := filepath.Clean(bundleRoot)
relative, err := filepath.Rel(root, bundle)
if err != nil {
return "", fmt.Errorf("compare notarius output paths: %w", err)
}
if relative == "." || relative == ".." || strings.HasPrefix(relative, ".."+string(filepath.Separator)) {
return "", fmt.Errorf("notarius output directory %q is not beneath output root %q", bundleRoot, outputRoot)
}
if err := requireDirectoryTree(root, relative); err != nil {
return "", fmt.Errorf("validate notarius output directory: %w", err)
}
return bundle, nil
}
func resolveRegularFile(root, logicalPath string) (string, error) {
resolved, err := pathsafe.JoinSlashRelativeUnderRoot(root, logicalPath)
if err != nil {
return "", err
}
relative, err := filepath.Rel(root, resolved)
if err != nil {
return "", err
}
if err := requireRegularFileTree(root, relative); err != nil {
return "", err
}
return resolved, nil
}
func requireDirectoryTree(root, relative string) error {
if err := requireDirectory(root); err != nil {
return err
}
current := root
for _, component := range strings.Split(relative, string(filepath.Separator)) {
current = filepath.Join(current, component)
if err := requireDirectory(current); err != nil {
return err
}
}
return nil
}
func requireRegularFileTree(root, relative string) error {
components := strings.Split(relative, string(filepath.Separator))
if len(components) == 0 {
return fmt.Errorf("regular file path is required")
}
if err := requireDirectory(root); err != nil {
return err
}
current := root
for _, component := range components[:len(components)-1] {
current = filepath.Join(current, component)
if err := requireDirectory(current); err != nil {
return err
}
}
return requireRegularFile(filepath.Join(current, components[len(components)-1]))
}
func requireDirectory(path string) error {
info, err := os.Lstat(path)
if err != nil {
return err
}
if info.Mode()&os.ModeSymlink != 0 || !info.IsDir() {
return fmt.Errorf("path %q must be a directory without symlinks", path)
}
return nil
}
func requireRegularFile(path string) error {
info, err := os.Lstat(path)
if err != nil {
return err
}
if info.Mode()&os.ModeSymlink != 0 || !info.Mode().IsRegular() {
return fmt.Errorf("path %q must be a regular file without symlinks", path)
}
return nil
}
func validateLogDestination(path string) error {
if err := requireDirectory(filepath.Dir(path)); err != nil {
return err
}
info, err := os.Lstat(path)
if errors.Is(err, os.ErrNotExist) {
return nil
}
if err != nil {
return err
}
if info.Mode()&os.ModeSymlink != 0 || !info.Mode().IsRegular() {
return fmt.Errorf("path %q must be absent or a regular file without symlinks", path)
}
return nil
}

View File

@@ -0,0 +1,572 @@
package notarius
import (
"context"
"encoding/json"
"errors"
"os"
"path/filepath"
"reflect"
"strings"
"testing"
"time"
sharedsubprocess "gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
)
func TestSubprocessRunnerBuildsExactInvocationAndDiscoversBundle(t *testing.T) {
req := validRunRequest(t)
var captured sharedsubprocess.RunRequest
runner := &SubprocessRunner{run: func(_ context.Context, processReq sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
captured = processReq
writeValidBundleAndReceipt(t, req, true)
return sharedsubprocess.RunResult{ExitCode: 0, Duration: 2 * time.Second}, nil
}}
result, err := runner.Run(context.Background(), req)
if err != nil {
t.Fatalf("Run() error = %v", err)
}
wantArgs := []string{
"run", "dnd-session", "--config", req.ConfigPath, "--input", req.InputPath,
"--output-dir", req.OutputRoot, "--json",
}
if !reflect.DeepEqual(captured.Args, wantArgs) {
t.Fatalf("subprocess args = %#v, want %#v", captured.Args, wantArgs)
}
if captured.Executable != req.Binary || captured.WorkingDir != req.WorkingDirectory || captured.Timeout != req.Timeout {
t.Fatalf("subprocess request = %#v", captured)
}
if captured.StdoutLogPath != req.ReceiptPath || captured.StderrLogPath != req.LogPath {
t.Fatalf("stream paths = stdout %q stderr %q", captured.StdoutLogPath, captured.StderrLogPath)
}
if captured.EnvOverrides != nil {
t.Fatalf("environment overrides = %#v, want inherited environment only", captured.EnvOverrides)
}
for _, arg := range captured.Args {
if arg == "--session-id" {
t.Fatal("subprocess args unexpectedly contain --session-id")
}
}
if result.Receipt.SchemaVersion != ReceiptSchemaVersion || result.Receipt.RunID != "notarius-run-1" {
t.Fatalf("receipt = %#v", result.Receipt)
}
if len(result.Index.Lanes) != 1 || result.Index.Lanes[0].LaneID != "npc-registry" {
t.Fatalf("lanes = %#v", result.Index.Lanes)
}
if result.Index.ChunkMap == nil || result.Index.ChunkMap.ArtifactKind != "chunk_map" {
t.Fatalf("chunk map = %#v", result.Index.ChunkMap)
}
if result.Index.EvidenceContext == nil || result.Index.EvidenceContext.ArtifactKind != "evidence_context" {
t.Fatalf("evidence context = %#v", result.Index.EvidenceContext)
}
if len(result.Rejections) != 1 || result.Rejections[0].LaneID != "spells" || result.Rejections[0].ReasonCode != "invalid_spell" {
t.Fatalf("rejections = %#v", result.Rejections)
}
if len(result.Warnings) != 1 || result.Warnings[0].Scope != "lane:npc-registry" || result.Warnings[0].ReasonCode != "normalized_name" {
t.Fatalf("warnings = %#v", result.Warnings)
}
}
func TestSubprocessRunnerInheritsEnvironmentAndSeparatesStreams(t *testing.T) {
req := validRunRequest(t)
writeValidBundleAndReceipt(t, req, false)
receiptFixture := req.ReceiptPath + ".fixture"
data, err := os.ReadFile(req.ReceiptPath)
if err != nil {
t.Fatalf("ReadFile(receipt) error = %v", err)
}
if err := os.WriteFile(receiptFixture, data, 0o644); err != nil {
t.Fatalf("WriteFile(receipt fixture) error = %v", err)
}
if err := os.Remove(req.ReceiptPath); err != nil {
t.Fatalf("Remove(receipt) error = %v", err)
}
captureDir := filepath.Join(filepath.Dir(req.ReceiptPath), "capture")
if err := os.Mkdir(captureDir, 0o755); err != nil {
t.Fatalf("Mkdir(capture) error = %v", err)
}
script := writeShellScript(t, `#!/bin/sh
pwd > "$NOTARIUS_CAPTURE_DIR/working-directory"
printf '%s' "$NOTARIUS_INHERITED_VALUE" > "$NOTARIUS_CAPTURE_DIR/environment"
printf 'diagnostic stream\n' >&2
cat "$NOTARIUS_RECEIPT_FIXTURE"
`)
req.Binary = script
t.Setenv("NOTARIUS_CAPTURE_DIR", captureDir)
t.Setenv("NOTARIUS_INHERITED_VALUE", "inherited-value")
t.Setenv("NOTARIUS_RECEIPT_FIXTURE", receiptFixture)
if _, err := NewSubprocessRunner().Run(context.Background(), req); err != nil {
t.Fatalf("Run() error = %v", err)
}
assertTextFile(t, filepath.Join(captureDir, "working-directory"), req.WorkingDirectory+"\n")
assertTextFile(t, filepath.Join(captureDir, "environment"), "inherited-value")
assertTextFile(t, req.LogPath, "diagnostic stream\n")
receiptBytes, err := os.ReadFile(req.ReceiptPath)
if err != nil {
t.Fatalf("ReadFile(receipt) error = %v", err)
}
if strings.Contains(string(receiptBytes), "diagnostic stream") {
t.Fatal("receipt contains stderr output")
}
}
func TestSubprocessRunnerReturnsProcessFailuresWithoutParsingStdout(t *testing.T) {
tests := []struct {
name string
scriptBody string
timeout time.Duration
cancel bool
want string
}{
{name: "nonzero", scriptBody: "printf '{malformed receipt'; printf 'failed\\n' >&2; exit 7\n", timeout: time.Second, want: "exit code 7"},
{name: "timeout", scriptBody: "sleep 5\n", timeout: 20 * time.Millisecond, want: "timed out"},
{name: "cancellation", scriptBody: "sleep 5\n", timeout: time.Second, cancel: true, want: "canceled"},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
req := validRunRequest(t)
req.Binary = writeShellScript(t, "#!/bin/sh\n"+test.scriptBody)
req.Timeout = test.timeout
ctx := context.Background()
if test.cancel {
cancelCtx, cancel := context.WithCancel(ctx)
ctx = cancelCtx
time.AfterFunc(20*time.Millisecond, cancel)
}
_, err := NewSubprocessRunner().Run(ctx, req)
if err == nil || !strings.Contains(err.Error(), test.want) {
t.Fatalf("Run() error = %v, want fragment %q", err, test.want)
}
if strings.Contains(err.Error(), "decode notarius receipt") {
t.Fatalf("Run() parsed stdout after process failure: %v", err)
}
})
}
}
func TestSubprocessRunnerReturnsSharedSubprocessErrorWithoutReadingReceipt(t *testing.T) {
req := validRunRequest(t)
if err := os.WriteFile(req.ReceiptPath, []byte("not json"), 0o644); err != nil {
t.Fatalf("WriteFile(receipt) error = %v", err)
}
wantErr := errors.New("process failed")
runner := &SubprocessRunner{run: func(context.Context, sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
return sharedsubprocess.RunResult{ExitCode: 9}, wantErr
}}
_, err := runner.Run(context.Background(), req)
if !errors.Is(err, wantErr) {
t.Fatalf("Run() error = %v, want wrapped process error", err)
}
if strings.Contains(err.Error(), "decode") {
t.Fatalf("Run() parsed receipt after failure: %v", err)
}
}
func TestLoadReceiptValidation(t *testing.T) {
root := t.TempDir()
valid := map[string]any{
"schema_version": ReceiptSchemaVersion, "run_id": "run-1", "pipeline_id": "pipeline-1",
"output_directory": filepath.Join(root, "outputs", "run-1"), "index_file": "index.json",
"normalized_output_count": 1, "rejected_output_count": 0, "warning_count": 0,
"validation_status": "approved", "future_field": true,
}
tests := []struct {
name string
mutate func(map[string]any)
raw []byte
wantOK bool
wantError string
}{
{name: "unknown fields tolerated", wantOK: true},
{name: "malformed", raw: []byte("{")},
{name: "unsupported version", mutate: func(v map[string]any) { v["schema_version"] = "notarius.run-result.v2" }},
{name: "missing field", mutate: func(v map[string]any) { delete(v, "run_id") }},
{name: "pipeline mismatch", mutate: func(v map[string]any) { v["pipeline_id"] = "other" }},
{name: "relative output", mutate: func(v map[string]any) { v["output_directory"] = "run-1" }},
{name: "negative count", mutate: func(v map[string]any) { v["warning_count"] = -1 }},
{
name: "nested index", mutate: func(v map[string]any) { v["index_file"] = "nested/index.json" },
wantError: `index_file "nested/index.json"`,
},
{
name: "cleanable index", mutate: func(v map[string]any) { v["index_file"] = "./index.json" },
wantError: `index_file "./index.json"`,
},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
path := filepath.Join(root, strings.ReplaceAll(test.name, " ", "-")+".json")
values := cloneMap(valid)
if test.mutate != nil {
test.mutate(values)
}
if test.raw != nil {
if err := os.WriteFile(path, test.raw, 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
} else {
writeJSONFile(t, path, values)
}
_, err := loadReceipt(path, "pipeline-1")
if test.wantOK && err != nil {
t.Fatalf("loadReceipt() error = %v", err)
}
if !test.wantOK && err == nil {
t.Fatal("loadReceipt() error = nil, want validation failure")
}
if test.wantError != "" && !strings.Contains(err.Error(), test.wantError) {
t.Fatalf("loadReceipt() error = %v, want fragment %q", err, test.wantError)
}
})
}
oversized := filepath.Join(root, "oversized.json")
if err := os.WriteFile(oversized, []byte(strings.Repeat("x", maxReceiptBytes+1)), 0o644); err != nil {
t.Fatalf("WriteFile(oversized) error = %v", err)
}
if _, err := loadReceipt(oversized, "pipeline-1"); err == nil || !strings.Contains(err.Error(), "exceeds") {
t.Fatalf("loadReceipt(oversized) error = %v", err)
}
}
func TestValidateBundleRootRejectsEscapesAndSymlinks(t *testing.T) {
root := t.TempDir()
outputRoot := filepath.Join(root, "output")
if err := os.Mkdir(outputRoot, 0o755); err != nil {
t.Fatalf("Mkdir(output root) error = %v", err)
}
validBundle := filepath.Join(outputRoot, "run-1")
if err := os.Mkdir(validBundle, 0o755); err != nil {
t.Fatalf("Mkdir(bundle) error = %v", err)
}
if _, err := validateBundleRoot(outputRoot, validBundle); err != nil {
t.Fatalf("validateBundleRoot(valid) error = %v", err)
}
outside := filepath.Join(root, "output-other")
if err := os.Mkdir(outside, 0o755); err != nil {
t.Fatalf("Mkdir(outside) error = %v", err)
}
for name, candidate := range map[string]string{"equal root": outputRoot, "escape": root, "prefix confusion": outside} {
t.Run(name, func(t *testing.T) {
if _, err := validateBundleRoot(outputRoot, candidate); err == nil {
t.Fatalf("validateBundleRoot(%q) error = nil", candidate)
}
})
}
symlink := filepath.Join(outputRoot, "linked")
if err := os.Symlink(outside, symlink); err != nil {
t.Skipf("Symlink() unavailable: %v", err)
}
if _, err := validateBundleRoot(outputRoot, symlink); err == nil {
t.Fatal("validateBundleRoot(symlink) error = nil")
}
}
func TestLoadIndexRejectsMalformedUnsafeAndUnsupportedDocuments(t *testing.T) {
tests := []struct {
name string
indexValue any
prepare func(*testing.T, string)
wantError string
}{
{name: "malformed", indexValue: json.RawMessage(`{"manifest_file":`)},
{name: "unsupported output shape", indexValue: map[string]any{"manifest_file": "manifest.json", "output_files": map[string]any{}, "rejected_file": "rejected.json", "warnings_file": "warnings.json"}},
{name: "missing management path", indexValue: map[string]any{"output_files": []any{}, "rejected_file": "rejected.json", "warnings_file": "warnings.json"}},
{name: "renamed manifest", indexValue: func() any {
value := validIndexValue([]any{})
value["manifest_file"] = "metadata.json"
return value
}(), wantError: `manifest_file "metadata.json"`},
{name: "cleanable manifest", indexValue: func() any {
value := validIndexValue([]any{})
value["manifest_file"] = "./manifest.json"
return value
}(), wantError: `manifest_file "./manifest.json"`},
{name: "renamed rejections", indexValue: func() any {
value := validIndexValue([]any{})
value["rejected_file"] = "rejections.json"
return value
}(), wantError: `rejected_file "rejections.json"`},
{name: "renamed warnings", indexValue: func() any {
value := validIndexValue([]any{})
value["warnings_file"] = "diagnostics/warnings.json"
return value
}(), wantError: `warnings_file "diagnostics/warnings.json"`},
{name: "duplicate lane", indexValue: validIndexValue([]any{
map[string]any{"lane_id": "npc", "file": "lanes/npc.json"},
map[string]any{"lane_id": "npc", "file": "lanes/npc.json"},
})},
{name: "absolute logical path", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "/tmp/npc.json"}})},
{name: "lexical traversal", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "../outside.json"}})},
{name: "root prefix confusion", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "../bundle-other/npc.json"}})},
{name: "file symlink", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "lanes/npc.json"}}), prepare: func(t *testing.T, bundle string) {
if err := os.Symlink(filepath.Join(bundle, "manifest.json"), filepath.Join(bundle, "lanes", "npc.json")); err != nil {
t.Skipf("Symlink() unavailable: %v", err)
}
}},
{name: "directory symlink", indexValue: validIndexValue([]any{map[string]any{"lane_id": "npc", "file": "linked/npc.json"}}), prepare: func(t *testing.T, bundle string) {
if err := os.Symlink(filepath.Join(bundle, "lanes"), filepath.Join(bundle, "linked")); err != nil {
t.Skipf("Symlink() unavailable: %v", err)
}
}},
{name: "missing management file", indexValue: validIndexValue([]any{}), prepare: func(t *testing.T, bundle string) {
if err := os.Remove(filepath.Join(bundle, "manifest.json")); err != nil {
t.Fatalf("Remove(manifest) error = %v", err)
}
}},
{name: "incomplete pipeline descriptor", indexValue: func() any {
value := validIndexValue([]any{})
value["chunk_map"] = map[string]any{"artifact_kind": "chunk_map", "file": "chunk-map.json"}
return value
}()},
{name: "pipeline descriptor escape", indexValue: func() any {
value := validIndexValue([]any{})
value["evidence_context"] = map[string]any{
"artifact_kind": "evidence_context", "file": "../evidence.json", "media_type": "application/json",
"schema_id": "evidence", "schema_name": "Evidence", "schema_version": "v1",
}
return value
}()},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
bundle := createBundleSkeleton(t)
indexPath := filepath.Join(bundle, "index.json")
if raw, ok := test.indexValue.(json.RawMessage); ok {
if err := os.WriteFile(indexPath, raw, 0o644); err != nil {
t.Fatalf("WriteFile(index) error = %v", err)
}
} else {
writeJSONFile(t, indexPath, test.indexValue)
}
if test.prepare != nil {
test.prepare(t, bundle)
}
if _, err := loadIndex(bundle, indexPath); err == nil {
t.Fatal("loadIndex() error = nil, want failure")
} else if test.wantError != "" && !strings.Contains(err.Error(), test.wantError) {
t.Fatalf("loadIndex() error = %v, want fragment %q", err, test.wantError)
}
})
}
bundle := createBundleSkeleton(t)
oversizedIndex := filepath.Join(bundle, "index.json")
if err := os.WriteFile(oversizedIndex, []byte(strings.Repeat("x", maxIndexBytes+1)), 0o644); err != nil {
t.Fatalf("WriteFile(oversized index) error = %v", err)
}
if _, err := loadIndex(bundle, oversizedIndex); err == nil || !strings.Contains(err.Error(), "exceeds") {
t.Fatalf("loadIndex(oversized) error = %v", err)
}
}
func TestLoadDiagnosticSummariesValidateBoundsAndTolerateUnknownFields(t *testing.T) {
root := t.TempDir()
rejectedPath := filepath.Join(root, "rejected.json")
warningsPath := filepath.Join(root, "warnings.json")
writeJSONFile(t, rejectedPath, map[string]any{"rejected": []any{map[string]any{
"stage": "validate", "lane_id": "spells", "reason_code": "invalid", "message": "do not retain this", "future": true,
}}, "future": true})
writeJSONFile(t, warningsPath, map[string]any{"warnings": []any{map[string]any{
"scope": "lane:spells", "reason_code": "bounded", "message": "do not retain this", "future": true,
}}, "future": true})
rejections, err := loadRejections(rejectedPath)
if err != nil || len(rejections) != 1 || rejections[0].ReasonCode != "invalid" {
t.Fatalf("loadRejections() = %#v, %v", rejections, err)
}
warnings, err := loadWarnings(warningsPath)
if err != nil || len(warnings) != 1 || warnings[0].Scope != "lane:spells" {
t.Fatalf("loadWarnings() = %#v, %v", warnings, err)
}
for name, path := range map[string]string{"rejections": rejectedPath, "warnings": warningsPath} {
t.Run("malformed "+name, func(t *testing.T) {
if err := os.WriteFile(path, []byte("{"), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
var err error
if name == "rejections" {
_, err = loadRejections(path)
} else {
_, err = loadWarnings(path)
}
if err == nil {
t.Fatal("summary decoder error = nil")
}
})
}
oversized := filepath.Join(root, "oversized.json")
if err := os.WriteFile(oversized, []byte(strings.Repeat("x", maxSummaryBytes+1)), 0o644); err != nil {
t.Fatalf("WriteFile(oversized) error = %v", err)
}
if _, err := loadWarnings(oversized); err == nil || !strings.Contains(err.Error(), "exceeds") {
t.Fatalf("loadWarnings(oversized) error = %v", err)
}
if _, err := loadRejections(oversized); err == nil || !strings.Contains(err.Error(), "exceeds") {
t.Fatalf("loadRejections(oversized) error = %v", err)
}
}
func TestFakeRunnerCapturesRequestsAndHonorsContextAndError(t *testing.T) {
req := RunRequest{PipelineID: "pipeline"}
want := RunResult{BundleRoot: "/bundle"}
fake := &FakeRunner{Result: want}
got, err := fake.Run(context.Background(), req)
if err != nil || !reflect.DeepEqual(got, want) || !reflect.DeepEqual(fake.Requests, []RunRequest{req}) {
t.Fatalf("Run() = %#v, %v; requests = %#v", got, err, fake.Requests)
}
wantErr := errors.New("configured failure")
fake.Err = wantErr
if _, err := fake.Run(context.Background(), req); !errors.Is(err, wantErr) {
t.Fatalf("Run(configured error) = %v", err)
}
canceled, cancel := context.WithCancel(context.Background())
cancel()
before := len(fake.Requests)
if _, err := fake.Run(canceled, req); !errors.Is(err, context.Canceled) || len(fake.Requests) != before {
t.Fatalf("Run(canceled) error = %v; requests = %d", err, len(fake.Requests))
}
}
func validRunRequest(t *testing.T) RunRequest {
t.Helper()
root := t.TempDir()
configPath := filepath.Join(root, "notarius.yml")
inputPath := filepath.Join(root, "input.json")
outputRoot := filepath.Join(root, "outputs")
workingDirectory := filepath.Join(root, "work")
diagnostics := filepath.Join(root, "diagnostics")
for _, directory := range []string{outputRoot, workingDirectory, diagnostics} {
if err := os.Mkdir(directory, 0o755); err != nil {
t.Fatalf("Mkdir(%q) error = %v", directory, err)
}
}
if err := os.WriteFile(configPath, []byte("pipelines: {}\n"), 0o644); err != nil {
t.Fatalf("WriteFile(config) error = %v", err)
}
if err := os.WriteFile(inputPath, []byte("{}\n"), 0o644); err != nil {
t.Fatalf("WriteFile(input) error = %v", err)
}
return RunRequest{
Binary: "notarius", ConfigPath: configPath, PipelineID: "dnd-session", InputPath: inputPath,
OutputRoot: outputRoot, WorkingDirectory: workingDirectory,
ReceiptPath: filepath.Join(diagnostics, "receipt.json"), LogPath: filepath.Join(diagnostics, "stderr.log"),
Timeout: time.Second,
}
}
func writeValidBundleAndReceipt(t *testing.T, req RunRequest, includeUnknown bool) {
t.Helper()
bundle := filepath.Join(req.OutputRoot, "notarius-run-1")
if err := os.MkdirAll(filepath.Join(bundle, "lanes"), 0o755); err != nil {
t.Fatalf("MkdirAll(bundle) error = %v", err)
}
for path, data := range map[string]string{
"manifest.json": `{}`,
"lanes/npc.json": `{}`,
"chunk-map.json": `{}`,
"evidence-context.json": `{}`,
} {
if err := os.WriteFile(filepath.Join(bundle, filepath.FromSlash(path)), []byte(data), 0o644); err != nil {
t.Fatalf("WriteFile(%q) error = %v", path, err)
}
}
rejection := map[string]any{"stage": "validate", "lane_id": "spells", "reason_code": "invalid_spell", "message": strings.Repeat("external detail", 20)}
warning := map[string]any{"scope": "lane:npc-registry", "reason_code": "normalized_name", "message": strings.Repeat("external warning", 20)}
if includeUnknown {
rejection["future"] = true
warning["future"] = true
}
writeJSONFile(t, filepath.Join(bundle, "rejected.json"), map[string]any{"rejected": []any{rejection}, "future": true})
writeJSONFile(t, filepath.Join(bundle, "warnings.json"), map[string]any{"warnings": []any{warning}, "future": true})
index := validIndexValue([]any{map[string]any{
"lane_id": "npc-registry", "file": "lanes/npc.json", "media_type": "application/json",
"module_key": "dnd/npc-registry", "schema_id": "notarius.dnd.npc_registry",
"schema_name": "NPCRegistry", "schema_version": "v1", "future": true,
}})
index["chunk_map"] = map[string]any{
"artifact_kind": "chunk_map", "file": "chunk-map.json", "media_type": "application/json",
"schema_id": "notarius.chunk_map", "schema_name": "ChunkMap", "schema_version": "v1", "future": true,
}
index["evidence_context"] = map[string]any{
"artifact_kind": "evidence_context", "file": "evidence-context.json", "media_type": "application/json",
"schema_id": "notarius.evidence_context", "schema_name": "EvidenceContext", "schema_version": "v1", "future": true,
}
index["future"] = true
writeJSONFile(t, filepath.Join(bundle, "index.json"), index)
receipt := map[string]any{
"schema_version": ReceiptSchemaVersion, "run_id": "notarius-run-1", "pipeline_id": req.PipelineID,
"output_directory": bundle, "index_file": "index.json", "normalized_output_count": 1,
"rejected_output_count": 1, "warning_count": 1, "validation_status": "rejected",
}
if includeUnknown {
receipt["future"] = true
}
writeJSONFile(t, req.ReceiptPath, receipt)
}
func createBundleSkeleton(t *testing.T) string {
t.Helper()
bundle := filepath.Join(t.TempDir(), "bundle")
if err := os.MkdirAll(filepath.Join(bundle, "lanes"), 0o755); err != nil {
t.Fatalf("MkdirAll(bundle) error = %v", err)
}
for _, name := range []string{"manifest.json", "rejected.json", "warnings.json", "lanes/npc.json", "chunk-map.json"} {
if err := os.WriteFile(filepath.Join(bundle, filepath.FromSlash(name)), []byte("{}"), 0o644); err != nil {
t.Fatalf("WriteFile(%q) error = %v", name, err)
}
}
return bundle
}
func validIndexValue(lanes []any) map[string]any {
return map[string]any{
"manifest_file": "manifest.json", "output_files": lanes,
"rejected_file": "rejected.json", "warnings_file": "warnings.json",
}
}
func writeJSONFile(t *testing.T, path string, value any) {
t.Helper()
data, err := json.Marshal(value)
if err != nil {
t.Fatalf("json.Marshal() error = %v", err)
}
if err := os.WriteFile(path, data, 0o644); err != nil {
t.Fatalf("WriteFile(%q) error = %v", path, err)
}
}
func writeShellScript(t *testing.T, body string) string {
t.Helper()
path := filepath.Join(t.TempDir(), "notarius-helper")
if err := os.WriteFile(path, []byte(body), 0o755); err != nil {
t.Fatalf("WriteFile(script) error = %v", err)
}
return path
}
func assertTextFile(t *testing.T, path, want string) {
t.Helper()
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("ReadFile(%q) error = %v", path, err)
}
if string(data) != want {
t.Fatalf("ReadFile(%q) = %q, want %q", path, string(data), want)
}
}
func cloneMap(source map[string]any) map[string]any {
result := make(map[string]any, len(source))
for key, value := range source {
result[key] = value
}
return result
}

View File

@@ -31,7 +31,7 @@ func TestSubprocessRunnerRunSuccessBuildsDeterministicArgsAndCapturesLogs(t *tes
ConfigPath: "/etc/scriptorium/config.yml",
PromptID: "dnd.session_recap",
ProfileID: "local-quality",
InputPaths: map[string]string{"transcript": filepath.Join(dir, "processed.json"), "other": filepath.Join(dir, "other.md")},
InputPaths: map[string]string{"transcript": filepath.Join(dir, "polished.json"), "other": filepath.Join(dir, "other.md")},
Vars: map[string]string{"session_id": "2026-05-03", "campaign_name": "Icewind Dale"},
OutputPath: filepath.Join(dir, "artifacts", "session_recap.md"),
StdoutLogPath: filepath.Join(dir, "logs", "scriptorium.run.stdout.log"),
@@ -180,7 +180,7 @@ func TestSubprocessRunnerRenderSuccess(t *testing.T) {
req := RenderArtifactRequest{
Binary: wrapper,
PromptID: "dnd.session_recap",
InputPaths: map[string]string{"transcript": filepath.Join(dir, "processed.json")},
InputPaths: map[string]string{"transcript": filepath.Join(dir, "polished.json")},
OutputPath: filepath.Join(dir, "artifacts", "session_recap.render.json"),
StdoutLogPath: filepath.Join(dir, "logs", "scriptorium.render.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "scriptorium.render.stderr.log"),
@@ -285,7 +285,7 @@ type scriptoriumHelperRecord struct {
func runReqForTest(t *testing.T, binary string) RunArtifactRequest {
t.Helper()
dir := t.TempDir()
transcriptPath := filepath.Join(dir, "processed.json")
transcriptPath := filepath.Join(dir, "polished.json")
writeScriptoriumFile(t, transcriptPath, `{"segments":[]}`)
return RunArtifactRequest{
Binary: binary,

View File

@@ -68,6 +68,26 @@ func (n *NoopRunner) Normalize(ctx context.Context, req NormalizeRequest) (Norma
}, nil
}
// Render returns the requested output path with placeholder metadata.
func (n *NoopRunner) Render(ctx context.Context, req RenderRequest) (RenderResult, error) {
if err := ctx.Err(); err != nil {
return RenderResult{}, err
}
if err := materializeRenderPlaceholders(req); err != nil {
return RenderResult{}, err
}
return RenderResult{
OutputRenderedPath: req.OutputRenderedPath,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
GeneratedConfigPath: req.GeneratedConfigPath,
InvokedBinary: "noop",
Format: req.Format,
Title: req.Title,
Metadata: map[string]any{"placeholder": true},
}, nil
}
// FakeRunner captures merge requests and returns deterministic responses.
type FakeRunner struct {
Requests []MergeRequest
@@ -79,6 +99,9 @@ type FakeRunner struct {
TrimRequests []TrimRequest
TrimErr error
TrimResult TrimResult
RenderRequests []RenderRequest
RenderErr error
RenderResult RenderResult
}
// Run records request and returns configured response.
@@ -195,6 +218,46 @@ func (f *FakeRunner) Normalize(ctx context.Context, req NormalizeRequest) (Norma
return res, nil
}
// Render records request and returns configured response.
func (f *FakeRunner) Render(ctx context.Context, req RenderRequest) (RenderResult, error) {
if err := ctx.Err(); err != nil {
return RenderResult{}, err
}
f.RenderRequests = append(f.RenderRequests, req)
if f.RenderErr != nil {
return RenderResult{}, f.RenderErr
}
if err := materializeRenderPlaceholders(req); err != nil {
return RenderResult{}, err
}
res := f.RenderResult
if res.OutputRenderedPath == "" {
res.OutputRenderedPath = req.OutputRenderedPath
}
if res.StdoutLogPath == "" {
res.StdoutLogPath = req.StdoutLogPath
}
if res.StderrLogPath == "" {
res.StderrLogPath = req.StderrLogPath
}
if res.GeneratedConfigPath == "" {
res.GeneratedConfigPath = req.GeneratedConfigPath
}
if res.InvokedBinary == "" {
res.InvokedBinary = "fake"
}
if res.Format == "" {
res.Format = req.Format
}
if res.Title == "" {
res.Title = req.Title
}
if res.Metadata == nil {
res.Metadata = map[string]any{"fake": true}
}
return res, nil
}
func materializePlaceholders(req MergeRequest) error {
if req.OutputMergedTranscriptPath != "" {
if err := subprocess.WriteFileAtomic(req.OutputMergedTranscriptPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), 0o644); err != nil {
@@ -301,3 +364,39 @@ func materializeNormalizePlaceholders(req NormalizeRequest) error {
}
return nil
}
func materializeRenderPlaceholders(req RenderRequest) error {
if req.OutputRenderedPath != "" {
if err := subprocess.WriteFileAtomic(req.OutputRenderedPath, []byte("# Transcript\n\nRendered markdown placeholder.\n"), 0o644); err != nil {
return fmt.Errorf("write rendered transcript %q: %w", req.OutputRenderedPath, err)
}
}
if req.GeneratedConfigPath != "" {
payload := map[string]any{
"schema": "seriatim.generated.v1",
"placeholder": true,
"command": "render",
"input_path": req.InputTranscriptPath,
"output_path": req.OutputRenderedPath,
"format": req.Format,
"title": req.Title,
"include_timestamps": req.IncludeTimestamps,
"include_segment_ids": req.IncludeSegmentIDs,
"include_metadata": req.IncludeMetadata,
}
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
}
}
if req.StdoutLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake render stdout placeholder\n"), 0o644); err != nil {
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
}
}
if req.StderrLogPath != "" {
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake render stderr placeholder\n"), 0o644); err != nil {
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
}
}
return nil
}

View File

@@ -14,7 +14,7 @@ func TestFakeRunnerCapturesRequestAndReturnsPath(t *testing.T) {
dir := t.TempDir()
req := MergeRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.yml"),
OutputMergedTranscriptPath: filepath.Join(dir, "transcripts", "merged.json"),
OutputMergedTranscriptPath: filepath.Join(dir, "transcripts", "base.json"),
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "seriatim.stderr.log"),
}
@@ -57,8 +57,8 @@ func TestFakeRunnerTrimCapturesRequestAndReturnsPath(t *testing.T) {
dir := t.TempDir()
req := TrimRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.trim.yml"),
InputTranscriptPath: filepath.Join(dir, "transcripts", "processed.json"),
OutputTrimmedPath: filepath.Join(dir, "transcripts", "trimmed.json"),
InputTranscriptPath: filepath.Join(dir, "transcripts", "polished.json"),
OutputTrimmedPath: filepath.Join(dir, "transcripts", "final.trimmed.json"),
KeepSelector: "1-10",
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.trim.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "seriatim.trim.stderr.log"),
@@ -105,8 +105,8 @@ func TestFakeRunnerNormalizeCapturesRequestAndReturnsPath(t *testing.T) {
dir := t.TempDir()
req := NormalizeRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.normalize.yml"),
InputTranscriptPath: filepath.Join(dir, "transcripts", "processed.json"),
OutputNormalizedPath: filepath.Join(dir, "transcripts", "normalized.json"),
InputTranscriptPath: filepath.Join(dir, "transcripts", "polished.json"),
OutputNormalizedPath: filepath.Join(dir, "transcripts", "final.json"),
OutputSchema: "seriatim-intermediate",
ReportPath: filepath.Join(dir, "artifacts", "seriatim.normalize.report.json"),
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.normalize.stdout.log"),
@@ -148,3 +148,58 @@ func TestFakeRunnerNormalizeError(t *testing.T) {
t.Fatal("expected error, got nil")
}
}
func TestFakeRunnerRenderCapturesRequestAndReturnsPath(t *testing.T) {
fake := &FakeRunner{}
dir := t.TempDir()
req := RenderRequest{
GeneratedConfigPath: filepath.Join(dir, "config", "seriatim.render.yml"),
InputTranscriptPath: filepath.Join(dir, "transcripts", "final.trimmed.json"),
OutputRenderedPath: filepath.Join(dir, "transcripts", "final.trimmed.md"),
Format: "markdown",
Title: "Session render",
IncludeTimestamps: true,
IncludeSegmentIDs: false,
IncludeMetadata: true,
StdoutLogPath: filepath.Join(dir, "logs", "seriatim.render.stdout.log"),
StderrLogPath: filepath.Join(dir, "logs", "seriatim.render.stderr.log"),
}
res, err := fake.Render(context.Background(), req)
if err != nil {
t.Fatalf("Render() error = %v", err)
}
if len(fake.RenderRequests) != 1 || fake.RenderRequests[0].GeneratedConfigPath == "" {
t.Fatalf("render requests = %#v, want captured request", fake.RenderRequests)
}
if res.OutputRenderedPath != req.OutputRenderedPath {
t.Fatalf("rendered path = %q, want %q", res.OutputRenderedPath, req.OutputRenderedPath)
}
if res.Format != req.Format {
t.Fatalf("format = %q, want %q", res.Format, req.Format)
}
if res.Title != req.Title {
t.Fatalf("title = %q, want %q", res.Title, req.Title)
}
cfgData, err := os.ReadFile(req.GeneratedConfigPath)
if err != nil {
t.Fatalf("read generated config: %v", err)
}
if !strings.Contains(string(cfgData), "command: render") {
t.Fatalf("generated config = %q, want render command marker", string(cfgData))
}
for _, path := range []string{req.StdoutLogPath, req.StderrLogPath, req.OutputRenderedPath} {
if _, err := os.Stat(path); err != nil {
t.Fatalf("expected file %q to exist: %v", path, err)
}
}
}
func TestFakeRunnerRenderError(t *testing.T) {
fake := &FakeRunner{RenderErr: errors.New("boom")}
_, err := fake.Render(context.Background(), RenderRequest{})
if err == nil {
t.Fatal("expected error, got nil")
}
}

View File

@@ -1,4 +1,4 @@
// Package seriatim declares the adapter contract for transcript merge/normalize/trim execution.
// Package seriatim declares the adapter contract for transcript merge/normalize/trim/render execution.
package seriatim
import (
@@ -6,11 +6,12 @@ import (
"time"
)
// Runner is the adapter boundary for seriatim merge/normalize/trim invocations.
// Runner is the adapter boundary for seriatim merge/normalize/trim/render invocations.
type Runner interface {
Run(ctx context.Context, req MergeRequest) (MergeResult, error)
Normalize(ctx context.Context, req NormalizeRequest) (NormalizeResult, error)
Trim(ctx context.Context, req TrimRequest) (TrimResult, error)
Render(ctx context.Context, req RenderRequest) (RenderResult, error)
}
// MergeRequest describes a seriatim merge invocation.
@@ -90,3 +91,33 @@ type TrimResult struct {
KeepSelector string
Metadata map[string]any
}
// RenderRequest describes a seriatim render invocation.
type RenderRequest struct {
Binary string
InputTranscriptPath string
OutputRenderedPath string
Format string
Title string
IncludeTimestamps bool
IncludeSegmentIDs bool
IncludeMetadata bool
StdoutLogPath string
StderrLogPath string
GeneratedConfigPath string
Timeout time.Duration
}
// RenderResult describes a render output.
type RenderResult struct {
OutputRenderedPath string
StdoutLogPath string
StderrLogPath string
GeneratedConfigPath string
ExitCode int
Duration time.Duration
InvokedBinary string
Format string
Title string
Metadata map[string]any
}

View File

@@ -8,6 +8,7 @@ import (
"strconv"
"strings"
"time"
"unicode/utf8"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
)
@@ -384,6 +385,96 @@ func (r *SubprocessRunner) Normalize(ctx context.Context, req NormalizeRequest)
}, nil
}
// Render executes Seriatim render with deterministic flags and validates non-empty text output.
func (r *SubprocessRunner) Render(ctx context.Context, req RenderRequest) (RenderResult, error) {
if r == nil {
return RenderResult{}, fmt.Errorf("seriatim subprocess runner is nil")
}
if strings.TrimSpace(req.InputTranscriptPath) == "" {
return RenderResult{}, fmt.Errorf("seriatim render input path is required")
}
if strings.TrimSpace(req.OutputRenderedPath) == "" {
return RenderResult{}, fmt.Errorf("seriatim render output path is required")
}
format := strings.TrimSpace(req.Format)
if format == "" {
format = "markdown"
}
if format != "markdown" {
return RenderResult{}, fmt.Errorf("seriatim render format %q is unsupported", req.Format)
}
binary := r.binary
if strings.TrimSpace(req.Binary) != "" {
binary = strings.TrimSpace(req.Binary)
}
timeout := r.timeout
if req.Timeout < 0 {
return RenderResult{}, fmt.Errorf("seriatim render timeout must be >= 0")
}
if req.Timeout > 0 {
timeout = req.Timeout
}
args := buildRenderArgs(req, format)
if req.GeneratedConfigPath != "" {
if err := writeRenderInvocationConfig(req, args, binary, timeout, format); err != nil {
return RenderResult{}, fmt.Errorf("write seriatim render invocation config %q: %w", req.GeneratedConfigPath, err)
}
}
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
Executable: binary,
Args: args,
Timeout: timeout,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
})
if err != nil {
return RenderResult{
OutputRenderedPath: req.OutputRenderedPath,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
GeneratedConfigPath: req.GeneratedConfigPath,
ExitCode: runRes.ExitCode,
Duration: runRes.Duration,
InvokedBinary: binary,
Format: format,
Title: req.Title,
}, fmt.Errorf("run seriatim render (binary=%q): %w", binary, err)
}
if err := validateNonEmptyTextFile(req.OutputRenderedPath); err != nil {
return RenderResult{
OutputRenderedPath: req.OutputRenderedPath,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
GeneratedConfigPath: req.GeneratedConfigPath,
ExitCode: runRes.ExitCode,
Duration: runRes.Duration,
InvokedBinary: binary,
Format: format,
Title: req.Title,
}, fmt.Errorf("validate seriatim rendered output %q: %w", req.OutputRenderedPath, err)
}
return RenderResult{
OutputRenderedPath: req.OutputRenderedPath,
StdoutLogPath: req.StdoutLogPath,
StderrLogPath: req.StderrLogPath,
GeneratedConfigPath: req.GeneratedConfigPath,
ExitCode: runRes.ExitCode,
Duration: runRes.Duration,
InvokedBinary: binary,
Format: format,
Title: req.Title,
Metadata: map[string]any{
"adapter": "seriatim_subprocess",
},
}, nil
}
func (r *SubprocessRunner) buildMergeArgs(req MergeRequest) []string {
args := []string{"merge"}
@@ -480,6 +571,22 @@ func buildNormalizeArgs(req NormalizeRequest, outputSchema string) []string {
return args
}
func buildRenderArgs(req RenderRequest, format string) []string {
args := []string{
"render",
"--input-file", req.InputTranscriptPath,
"--output-file", req.OutputRenderedPath,
"--format", format,
"--include-timestamps=" + strconv.FormatBool(req.IncludeTimestamps),
"--include-segment-ids=" + strconv.FormatBool(req.IncludeSegmentIDs),
"--include-metadata=" + strconv.FormatBool(req.IncludeMetadata),
}
if strings.TrimSpace(req.Title) != "" {
args = append(args, "--title", req.Title)
}
return args
}
func writeTrimInvocationConfig(req TrimRequest, args []string, binary string, timeout time.Duration) error {
payload := map[string]any{
"schema": "seriatim.generated.v1",
@@ -509,6 +616,24 @@ func writeNormalizeInvocationConfig(req NormalizeRequest, args []string, binary
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
}
func writeRenderInvocationConfig(req RenderRequest, args []string, binary string, timeout time.Duration, format string) error {
payload := map[string]any{
"schema": "seriatim.generated.v1",
"command": "render",
"binary": binary,
"args": args,
"timeout": timeout.String(),
"input_path": req.InputTranscriptPath,
"output_path": req.OutputRenderedPath,
"format": format,
"title": req.Title,
"include_timestamps": req.IncludeTimestamps,
"include_segment_ids": req.IncludeSegmentIDs,
"include_metadata": req.IncludeMetadata,
}
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
}
func validateJSONFile(path string) error {
data, err := os.ReadFile(path)
if err != nil {
@@ -541,3 +666,20 @@ func validateJSONFileWithSegments(path string) error {
}
return nil
}
func validateNonEmptyTextFile(path string) error {
data, err := os.ReadFile(path)
if err != nil {
return fmt.Errorf("read file: %w", err)
}
if len(data) == 0 {
return fmt.Errorf("file is empty")
}
if !utf8.Valid(data) {
return fmt.Errorf("file is not valid utf-8 text")
}
if strings.TrimSpace(string(data)) == "" {
return fmt.Errorf("file has no non-whitespace content")
}
return nil
}

View File

@@ -50,7 +50,7 @@ func TestSubprocessRunnerSuccessWithReportArgsAndEnv(t *testing.T) {
req := MergeRequest{
GeneratedConfigPath: filepath.Join(dir, "seriatim.generated.yml"),
InputTranscriptPaths: []string{filepath.Join(dir, "a.json"), filepath.Join(dir, "b.json")},
OutputMergedTranscriptPath: filepath.Join(dir, "merged.json"),
OutputMergedTranscriptPath: filepath.Join(dir, "base.json"),
ReportPath: filepath.Join(dir, "seriatim.report.json"),
SpeakersPath: filepath.Join(dir, "speakers.yml"),
AutocorrectPath: filepath.Join(dir, "autocorrect.yml"),
@@ -569,6 +569,156 @@ func TestSubprocessRunnerNormalizeInvalidReportJSONFails(t *testing.T) {
}
}
func TestSubprocessRunnerRenderSuccessInvocationAndProvenance(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
t.Setenv("SERIATIM_HELPER_MODE", "render_success")
recordPath := filepath.Join(t.TempDir(), "record.json")
t.Setenv("SERIATIM_HELPER_RECORD_PATH", recordPath)
wrapper := writeHelperWrapper(t)
runner := mustRunner(t, wrapper, false)
req := renderReqForTest(t)
res, err := runner.Render(context.Background(), req)
if err != nil {
t.Fatalf("Render() error = %v", err)
}
if res.OutputRenderedPath != req.OutputRenderedPath {
t.Fatalf("OutputRenderedPath = %q, want %q", res.OutputRenderedPath, req.OutputRenderedPath)
}
if res.Format != req.Format {
t.Fatalf("Format = %q, want %q", res.Format, req.Format)
}
if res.Title != req.Title {
t.Fatalf("Title = %q, want %q", res.Title, req.Title)
}
if res.InvokedBinary != wrapper {
t.Fatalf("InvokedBinary = %q, want %q", res.InvokedBinary, wrapper)
}
if res.ExitCode != 0 {
t.Fatalf("ExitCode = %d, want 0", res.ExitCode)
}
if res.Duration <= 0 {
t.Fatalf("Duration = %s, want >0", res.Duration)
}
if res.Metadata == nil || res.Metadata["adapter"] != "seriatim_subprocess" {
t.Fatalf("Metadata = %#v, want adapter marker", res.Metadata)
}
if _, err := os.Stat(req.OutputRenderedPath); err != nil {
t.Fatalf("rendered output missing: %v", err)
}
if _, err := os.Stat(req.StdoutLogPath); err != nil {
t.Fatalf("stdout log missing: %v", err)
}
if _, err := os.Stat(req.StderrLogPath); err != nil {
t.Fatalf("stderr log missing: %v", err)
}
if _, err := os.Stat(req.GeneratedConfigPath); err != nil {
t.Fatalf("generated config missing: %v", err)
}
rec := readHelperRecord(t, recordPath)
wantArgs := []string{
"render",
"--input-file", req.InputTranscriptPath,
"--output-file", req.OutputRenderedPath,
"--format", req.Format,
"--include-timestamps=true",
"--include-segment-ids=true",
"--include-metadata=false",
"--title", req.Title,
}
if strings.Join(rec.Args, "\n") != strings.Join(wantArgs, "\n") {
t.Fatalf("args = %#v, want %#v", rec.Args, wantArgs)
}
}
func TestSubprocessRunnerRenderWithoutTitleOmitsTitleArg(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
t.Setenv("SERIATIM_HELPER_MODE", "render_success")
recordPath := filepath.Join(t.TempDir(), "record.json")
t.Setenv("SERIATIM_HELPER_RECORD_PATH", recordPath)
runner := mustRunner(t, writeHelperWrapper(t), false)
req := renderReqForTest(t)
req.Title = ""
if _, err := runner.Render(context.Background(), req); err != nil {
t.Fatalf("Render() error = %v", err)
}
rec := readHelperRecord(t, recordPath)
for i := 0; i < len(rec.Args); i++ {
if rec.Args[i] == "--title" {
t.Fatalf("args = %#v, did not expect --title", rec.Args)
}
}
}
func TestSubprocessRunnerRenderSubprocessFailure(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
t.Setenv("SERIATIM_HELPER_MODE", "fail")
t.Setenv("SERIATIM_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
runner := mustRunner(t, writeHelperWrapper(t), false)
req := renderReqForTest(t)
_, err := runner.Render(context.Background(), req)
if err == nil {
t.Fatal("Render() error = nil, want non-nil")
}
if !strings.Contains(err.Error(), "run seriatim render") {
t.Fatalf("error = %q, want subprocess context", err.Error())
}
}
func TestSubprocessRunnerRenderMissingOutputFails(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
t.Setenv("SERIATIM_HELPER_MODE", "missing_output")
t.Setenv("SERIATIM_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
runner := mustRunner(t, writeHelperWrapper(t), false)
req := renderReqForTest(t)
_, err := runner.Render(context.Background(), req)
if err == nil {
t.Fatal("Render() error = nil, want non-nil")
}
if !strings.Contains(err.Error(), "validate seriatim rendered output") {
t.Fatalf("error = %q, want output validation context", err.Error())
}
}
func TestSubprocessRunnerRenderEmptyOutputFails(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_SERIATIM_HELPER", "1")
t.Setenv("SERIATIM_HELPER_MODE", "render_empty_output")
t.Setenv("SERIATIM_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
runner := mustRunner(t, writeHelperWrapper(t), false)
req := renderReqForTest(t)
_, err := runner.Render(context.Background(), req)
if err == nil {
t.Fatal("Render() error = nil, want non-nil")
}
if !strings.Contains(err.Error(), "file is empty") {
t.Fatalf("error = %q, want empty-file validation", err.Error())
}
}
func TestSubprocessRunnerConstructorValidation(t *testing.T) {
_, err := NewSubprocessRunnerFromConfigValues("", "10m", "seriatim-intermediate", nil, true, EnvConfig{})
if err == nil {
@@ -702,6 +852,14 @@ func TestSeriatimSubprocessHelper(t *testing.T) {
case "normalize_report_missing":
writeSeriatimHelperFile(outputPath, `{"schema":"seriatim.intermediate.v1","segments":[]}`)
os.Exit(0)
case "render_success":
writeSeriatimHelperFile(outputPath, "# Rendered transcript\n\nHello.\n")
_, _ = os.Stdout.WriteString("seriatim helper render stdout\n")
_, _ = os.Stderr.WriteString("seriatim helper render stderr\n")
os.Exit(0)
case "render_empty_output":
writeSeriatimHelperFile(outputPath, "")
os.Exit(0)
default:
_, _ = os.Stderr.WriteString(fmt.Sprintf("unknown helper mode %q\n", mode))
os.Exit(2)
@@ -732,7 +890,7 @@ func mergeReqForTest(t *testing.T, withReport bool) MergeRequest {
req := MergeRequest{
GeneratedConfigPath: filepath.Join(dir, "seriatim.generated.yml"),
InputTranscriptPaths: []string{in1, in2},
OutputMergedTranscriptPath: filepath.Join(dir, "merged.json"),
OutputMergedTranscriptPath: filepath.Join(dir, "base.json"),
StdoutLogPath: filepath.Join(dir, "seriatim.stdout.log"),
StderrLogPath: filepath.Join(dir, "seriatim.stderr.log"),
}
@@ -745,11 +903,11 @@ func mergeReqForTest(t *testing.T, withReport bool) MergeRequest {
func trimReqForTest(t *testing.T) TrimRequest {
t.Helper()
dir := t.TempDir()
input := filepath.Join(dir, "processed.json")
input := filepath.Join(dir, "polished.json")
writeSeriatimFile(t, input, `{"schema":"seriatim.intermediate.v1","segments":[]}`)
return TrimRequest{
InputTranscriptPath: input,
OutputTrimmedPath: filepath.Join(dir, "trimmed.json"),
OutputTrimmedPath: filepath.Join(dir, "final.trimmed.json"),
KeepSelector: "5-12",
GeneratedConfigPath: filepath.Join(dir, "seriatim.trim.generated.yml"),
StdoutLogPath: filepath.Join(dir, "seriatim.trim.stdout.log"),
@@ -760,12 +918,12 @@ func trimReqForTest(t *testing.T) TrimRequest {
func normalizeReqForTest(t *testing.T, withReport bool) NormalizeRequest {
t.Helper()
dir := t.TempDir()
input := filepath.Join(dir, "processed.json")
input := filepath.Join(dir, "polished.json")
writeSeriatimFile(t, input, `{"schema":"audita.processed.v1","segments":[]}`)
req := NormalizeRequest{
InputTranscriptPath: input,
OutputNormalizedPath: filepath.Join(dir, "normalized.json"),
OutputNormalizedPath: filepath.Join(dir, "final.json"),
OutputSchema: "seriatim-intermediate",
GeneratedConfigPath: filepath.Join(dir, "seriatim.normalize.generated.yml"),
StdoutLogPath: filepath.Join(dir, "seriatim.normalize.stdout.log"),
@@ -777,6 +935,25 @@ func normalizeReqForTest(t *testing.T, withReport bool) NormalizeRequest {
return req
}
func renderReqForTest(t *testing.T) RenderRequest {
t.Helper()
dir := t.TempDir()
input := filepath.Join(dir, "final.trimmed.json")
writeSeriatimFile(t, input, `{"schema":"seriatim.intermediate.v1","segments":[]}`)
return RenderRequest{
InputTranscriptPath: input,
OutputRenderedPath: filepath.Join(dir, "final.trimmed.md"),
Format: "markdown",
Title: "Session 42",
IncludeTimestamps: true,
IncludeSegmentIDs: true,
IncludeMetadata: false,
GeneratedConfigPath: filepath.Join(dir, "seriatim.render.generated.yml"),
StdoutLogPath: filepath.Join(dir, "seriatim.render.stdout.log"),
StderrLogPath: filepath.Join(dir, "seriatim.render.stderr.log"),
}
}
func mustRunner(t *testing.T, binary string, report bool) *SubprocessRunner {
t.Helper()
coalesce := 3.0

View File

@@ -1,31 +0,0 @@
// Package storage declares archive/storage backend adapter boundaries.
package storage
import "context"
// TODO: implement remote storage/archive backends (S3/SFTP/etc.).
// Backend is the adapter boundary for archive/storage operations.
type Backend interface {
Archive(ctx context.Context, req ArchiveRequest) (ArchiveResult, error)
}
// ArchiveItem describes one item to archive.
type ArchiveItem struct {
Kind string
LocalPath string
RemoteKey string
}
// ArchiveRequest describes one archive operation.
type ArchiveRequest struct {
SessionID string
ManifestPath string
Items []ArchiveItem
}
// ArchiveResult describes archive operation output.
type ArchiveResult struct {
Archived []ArchiveItem
Metadata map[string]any
}

View File

@@ -10,25 +10,11 @@ import (
"time"
)
// NoopBackend is a deterministic no-op archive/storage adapter.
type NoopBackend struct{}
// Archive returns the requested items as archived with placeholder metadata.
func (n *NoopBackend) Archive(ctx context.Context, req ArchiveRequest) (ArchiveResult, error) {
if err := ctx.Err(); err != nil {
return ArchiveResult{}, err
}
return ArchiveResult{Archived: append([]ArchiveItem(nil), req.Items...), Metadata: map[string]any{"placeholder": true}}, nil
}
// FakeBackend captures archive requests and returns deterministic responses.
// FakeBackend provides a deterministic in-memory object store for tests.
type FakeBackend struct {
Requests []ArchiveRequest
Err error
Result ArchiveResult
Objects map[string]FakeObject
Uploads []FakeUploadCall
Objects map[string]FakeObject
Uploads []FakeUploadCall
Downloads []FakeDownloadCall
ListErr error
DownloadErr error
@@ -43,23 +29,10 @@ type FakeUploadCall struct {
Options UploadOptions
}
// Archive records request and returns configured response.
func (f *FakeBackend) Archive(ctx context.Context, req ArchiveRequest) (ArchiveResult, error) {
if err := ctx.Err(); err != nil {
return ArchiveResult{}, err
}
f.Requests = append(f.Requests, req)
if f.Err != nil {
return ArchiveResult{}, f.Err
}
res := f.Result
if res.Archived == nil {
res.Archived = append([]ArchiveItem(nil), req.Items...)
}
if res.Metadata == nil {
res.Metadata = map[string]any{"fake": true}
}
return res, nil
// FakeDownloadCall captures one download invocation in call order.
type FakeDownloadCall struct {
Key string
LocalPath string
}
// FakeObject is a deterministic fake object-store record.
@@ -130,6 +103,10 @@ func (f *FakeBackend) Download(ctx context.Context, key, localPath string) error
if !ok {
return fmt.Errorf("download object %q: %w", key, os.ErrNotExist)
}
f.Downloads = append(f.Downloads, FakeDownloadCall{
Key: normalizeObjectKey(key),
LocalPath: localPath,
})
if err := os.MkdirAll(filepath.Dir(localPath), 0o755); err != nil {
return fmt.Errorf("download object %q: create parent directory: %w", key, err)

View File

@@ -9,30 +9,6 @@ import (
"testing"
)
func TestFakeBackendCapturesRequestAndReturnsItems(t *testing.T) {
fake := &FakeBackend{}
req := ArchiveRequest{SessionID: "s1", Items: []ArchiveItem{{Kind: "artifact", LocalPath: "artifacts/log.md"}}}
res, err := fake.Archive(context.Background(), req)
if err != nil {
t.Fatalf("Archive() error = %v", err)
}
if len(fake.Requests) != 1 || fake.Requests[0].SessionID != "s1" {
t.Fatalf("requests = %#v, want captured request", fake.Requests)
}
if len(res.Archived) != 1 {
t.Fatalf("archived len = %d, want 1", len(res.Archived))
}
}
func TestFakeBackendError(t *testing.T) {
fake := &FakeBackend{Err: errors.New("boom")}
_, err := fake.Archive(context.Background(), ArchiveRequest{})
if err == nil {
t.Fatal("expected error, got nil")
}
}
func TestFakeBackendListPrefixFiltering(t *testing.T) {
fake := &FakeBackend{}
fake.SeedObject(FakeObject{Key: "dnd/campaigns/forsaken/audio/a.flac", Data: []byte("a")})
@@ -56,7 +32,7 @@ func TestFakeBackendDownload(t *testing.T) {
fake.SeedObject(FakeObject{Key: "audio/a.flac", Data: []byte("audio-a")})
dst := filepath.Join(t.TempDir(), "nested", "a.flac")
if err := fake.Download(context.Background(), "audio/a.flac", dst); err != nil {
if err := fake.Download(context.Background(), `audio\a.flac`, dst); err != nil {
t.Fatalf("Download() error = %v", err)
}
data, err := os.ReadFile(dst)

View File

@@ -5,7 +5,7 @@ import (
"time"
)
// ObjectStore is a remote object storage boundary used by future prepare/archive work.
// ObjectStore is a remote object storage boundary used by prepare, restore, and publish work.
//
// Key invariant:
// callers pass full bucket-relative object keys. Backend implementations do not

View File

@@ -0,0 +1,36 @@
package storage
import (
"context"
"fmt"
"os"
"path/filepath"
"strings"
)
// DownloadObjectToTemp downloads an object into a temporary file and returns
// the cleaned local path.
func DownloadObjectToTemp(ctx context.Context, store ObjectStore, key, pattern string) (string, error) {
if store == nil {
return "", fmt.Errorf("object store is required")
}
if strings.TrimSpace(pattern) == "" {
return "", fmt.Errorf("temp file pattern is required")
}
tmp, err := os.CreateTemp("", pattern)
if err != nil {
return "", fmt.Errorf("create temp file: %w", err)
}
path := tmp.Name()
if err := tmp.Close(); err != nil {
_ = os.Remove(path)
return "", fmt.Errorf("close temp file: %w", err)
}
if err := store.Download(ctx, key, path); err != nil {
_ = os.Remove(path)
return "", err
}
return filepath.Clean(path), nil
}

View File

@@ -0,0 +1,69 @@
package storage
import (
"context"
"errors"
"fmt"
"os"
"path/filepath"
"strings"
"testing"
)
func TestDownloadObjectToTempSuccess(t *testing.T) {
store := &FakeBackend{}
store.SeedObject(FakeObject{Key: "sessions/a/current/run_id.txt", Data: []byte("run-123\n")})
path, err := DownloadObjectToTemp(context.Background(), store, "sessions/a/current/run_id.txt", "narratio-test-*.txt")
if err != nil {
t.Fatalf("DownloadObjectToTemp() error = %v", err)
}
t.Cleanup(func() { _ = os.Remove(path) })
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("ReadFile() error = %v", err)
}
if string(data) != "run-123\n" {
t.Fatalf("downloaded data = %q, want %q", string(data), "run-123\n")
}
}
func TestDownloadObjectToTempFailedDownloadRemovesTempFile(t *testing.T) {
sentinel := errors.New("download failed")
store := &FakeBackend{DownloadErr: sentinel}
pattern := "narratio-test-fail-*.txt"
before, err := filepath.Glob(filepath.Join(os.TempDir(), "narratio-test-fail-*.txt"))
if err != nil {
t.Fatalf("Glob(before) error = %v", err)
}
path, err := DownloadObjectToTemp(context.Background(), store, "sessions/a/current/run_id.txt", pattern)
if !errors.Is(err, sentinel) {
t.Fatalf("DownloadObjectToTemp() error = %v, want %v", err, sentinel)
}
if strings.TrimSpace(path) != "" {
t.Fatalf("DownloadObjectToTemp() path = %q, want empty on failure", path)
}
after, err := filepath.Glob(filepath.Join(os.TempDir(), "narratio-test-fail-*.txt"))
if err != nil {
t.Fatalf("Glob(after) error = %v", err)
}
if len(after) != len(before) {
t.Fatalf("temp file count changed after failed download: before=%d after=%d", len(before), len(after))
}
}
func TestDownloadObjectToTempCallerContextWrappingPreservesCause(t *testing.T) {
sentinel := errors.New("object missing")
store := &FakeBackend{DownloadErr: sentinel}
_, err := DownloadObjectToTemp(context.Background(), store, "sessions/a/current/run_id.txt", "narratio-test-*.txt")
if err == nil {
t.Fatal("DownloadObjectToTemp() error = nil, want error")
}
err = fmt.Errorf("download run pointer failed: %w", err)
if !errors.Is(err, sentinel) {
t.Fatalf("wrapped error does not preserve sentinel cause: %v", err)
}
}

View File

@@ -145,7 +145,7 @@ func TestHTTPClientDoesNotRetryOnNonRetryableStatus(t *testing.T) {
}
}
func TestHTTPClientInvalidJSONFailsAndDoesNotPromote(t *testing.T) {
func TestHTTPClientInvalidJSONFailsAndDoesNotInstallOutput(t *testing.T) {
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
_, _ = w.Write([]byte(`not-json`))
}))

View File

@@ -46,7 +46,7 @@ func (f *artifactSelectionFlag) Normalize() ([]string, error) {
return out, nil
}
func validateSelectedAnalyzeArtifacts(cfg *config.Config, selected []string) error {
func validateSelectedArtifacts(cfg *config.Config, selected []string) error {
if len(selected) == 0 {
return nil
}

View File

@@ -9,36 +9,80 @@ import (
"testing"
"time"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
"gitea.maximumdirect.net/eric/narratio/internal/stage"
)
func TestExecuteRunStageArtifactsNonAnalyzeFails(t *testing.T) {
func TestExecuteRunStageArtifactsUnsupportedStageFails(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "--artifacts", "session_recap", "polish"},
[]string{"run-stage", "extract", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "session_recap"},
&stdout,
&stderr,
)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), `run-stage: --artifacts is only supported for stage "analyze"`) {
if !strings.Contains(stderr.String(), `run-stage: --artifacts is only supported for stages "analyze" and "publish"`) {
t.Fatalf("stderr = %q, want stage-gating error", stderr.String())
}
}
func TestExecuteRunStagePublishPropagatesSelectedArtifacts(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var capturedStages []string
var capturedArtifacts []string
origExecuteStagesFn := executeStagesFn
t.Cleanup(func() {
executeStagesFn = origExecuteStagesFn
})
executeStagesFn = func(_ context.Context, _ *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
for _, s := range stages {
capturedStages = append(capturedStages, s.Name())
}
capturedArtifacts = append([]string(nil), opts.SelectedArtifacts...)
return &RunSummary{ManifestPath: filepath.Join(workspaceRoot, "manifest.json"), Executed: []string{"publish"}}, nil
}
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{
"run-stage", "publish", "2026-05-03",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--artifacts", "session_recap",
},
&stdout,
&stderr,
)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if len(capturedStages) != 1 || capturedStages[0] != "publish" {
t.Fatalf("captured stages = %#v, want [publish]", capturedStages)
}
if strings.Join(capturedArtifacts, ",") != "session_recap" {
t.Fatalf("captured artifacts = %#v, want [session_recap]", capturedArtifacts)
}
}
func TestExecuteUnknownArtifactsFailValidation(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"run", "--config", pipelinePath, "--session", sessionPath, "--artifacts", "unknown_artifact"},
[]string{"run", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "unknown_artifact"},
&stdout,
&stderr,
)
@@ -52,7 +96,7 @@ func TestExecuteUnknownArtifactsFailValidation(t *testing.T) {
func TestRunStageArtifactsDoesNotImplyForce(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{}
@@ -65,7 +109,7 @@ func TestRunStageArtifactsDoesNotImplyForce(t *testing.T) {
var out bytes.Buffer
err := RunStage(
context.Background(),
[]string{"--config", pipelinePath, "--session", sessionPath, "--artifacts", "session_recap,session_recap", "analyze"},
[]string{"analyze", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "session_recap,session_recap"},
&out,
)
if err != nil {
@@ -76,38 +120,289 @@ func TestRunStageArtifactsDoesNotImplyForce(t *testing.T) {
}
}
func TestResumeArtifactsWithSucceededAnalyzeSkipsUnlessForced(t *testing.T) {
func TestRunArtifactsWithSucceededAnalyzeSkipsUnlessForced(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{}
seed := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
for _, stageName := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "analyze", "archive", "notify"} {
for _, stageName := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "render", "analyze", "publish", "notify"} {
seed.MarkStageSucceeded(stageName, time.Date(2026, 5, 3, 10, 1, 0, 0, time.UTC), nil)
}
seed.MarkStageSkipped("extract", time.Date(2026, 5, 3, 10, 1, 0, 0, time.UTC), "notarius_disabled")
if err := store.Save(context.Background(), manifestPath, seed); err != nil {
t.Fatalf("save manifest: %v", err)
}
var out bytes.Buffer
err := Resume(
err := Run(
context.Background(),
[]string{"--config", pipelinePath, "--session", sessionPath, "--artifacts", "session_recap"},
[]string{"2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "session_recap"},
&out,
)
if err != nil {
t.Fatalf("Resume() error = %v", err)
t.Fatalf("Run() error = %v", err)
}
if !strings.Contains(out.String(), "has no remaining stages") {
t.Fatalf("output = %q, want no remaining stages", out.String())
if !strings.Contains(out.String(), "executed=1 skipped=11") {
t.Fatalf("output = %q, want all stages skipped", out.String())
}
}
func writeValidConfigFilesWithScriptoriumArtifacts(t *testing.T, workspaceRoot string) (string, string) {
func TestExecuteAnalyzeForceRunsAnalyze(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var capturedStages []string
var capturedForce bool
origExecuteStagesFn := executeStagesFn
t.Cleanup(func() {
executeStagesFn = origExecuteStagesFn
})
executeStagesFn = func(_ context.Context, _ *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
for _, s := range stages {
capturedStages = append(capturedStages, s.Name())
}
capturedForce = opts.Force
return &RunSummary{
ManifestPath: filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json"),
Executed: []string{"analyze"},
}, nil
}
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"analyze", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath},
&stdout,
&stderr,
)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if len(capturedStages) != 1 || capturedStages[0] != "analyze" {
t.Fatalf("captured stages = %#v, want [analyze]", capturedStages)
}
if !capturedForce {
t.Fatal("captured force = false, want true")
}
if !strings.Contains(stdout.String(), "narratio analyze: executed=1 skipped=0 force=true; manifest=") {
t.Fatalf("stdout = %q, want analyze summary", stdout.String())
}
}
func TestExecuteAnalyzePropagatesSelectedArtifacts(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var capturedArtifacts []string
origExecuteStagesFn := executeStagesFn
t.Cleanup(func() {
executeStagesFn = origExecuteStagesFn
})
executeStagesFn = func(_ context.Context, _ *config.Config, _ []stage.Stage, opts RunOptions) (*RunSummary, error) {
capturedArtifacts = append([]string(nil), opts.SelectedArtifacts...)
return &RunSummary{ManifestPath: filepath.Join(workspaceRoot, "manifest.json"), Executed: []string{"analyze"}}, nil
}
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{
"analyze",
"2026-05-03",
"--config", pipelinePath,
"--campaign-file", campaignPath,
"--session", sessionPath,
"--artifacts", "player_handout,session_recap",
},
&stdout,
&stderr,
)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if strings.Join(capturedArtifacts, ",") != "player_handout,session_recap" {
t.Fatalf("captured artifacts = %#v, want sorted selected artifacts", capturedArtifacts)
}
}
func TestExecuteAnalyzeUnknownArtifactFailsValidation(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"analyze", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "unknown_artifact"},
&stdout,
&stderr,
)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), `analyze: --artifacts includes unknown artifact "unknown_artifact"`) {
t.Fatalf("stderr = %q, want unknown-artifact validation error", stderr.String())
}
}
func TestExecuteAnalyzeRejectsPositionalArgsAndForceFlag(t *testing.T) {
cases := []struct {
name string
args []string
want string
}{
{name: "extra positional", args: []string{"analyze", "2026-05-03", "extra"}, want: "analyze: unexpected positional arguments"},
{name: "force flag", args: []string{"analyze", "--force"}, want: "analyze: invalid flags: flag provided but not defined: -force"},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(tc.args, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), tc.want) {
t.Fatalf("stderr = %q, want %q", stderr.String(), tc.want)
}
})
}
}
func TestExecuteAnalyzeMissingConfigUsesRunStageLoadingPath(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"analyze", "2026-05-03"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "analyze: no pipeline config path provided and no default pipeline config found; searched:") {
t.Fatalf("stderr = %q, want pipeline discovery error", stderr.String())
}
}
func TestExecutePublishForceRunsPublish(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var capturedStages []string
var capturedForce bool
var capturedArtifacts []string
origExecuteStagesFn := executeStagesFn
t.Cleanup(func() {
executeStagesFn = origExecuteStagesFn
})
executeStagesFn = func(_ context.Context, _ *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
for _, s := range stages {
capturedStages = append(capturedStages, s.Name())
}
capturedForce = opts.Force
capturedArtifacts = append([]string(nil), opts.SelectedArtifacts...)
return &RunSummary{
ManifestPath: filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json"),
Executed: []string{"publish"},
}, nil
}
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"publish", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "session_recap"},
&stdout,
&stderr,
)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if len(capturedStages) != 1 || capturedStages[0] != "publish" {
t.Fatalf("captured stages = %#v, want [publish]", capturedStages)
}
if !capturedForce {
t.Fatal("captured force = false, want true")
}
if strings.Join(capturedArtifacts, ",") != "session_recap" {
t.Fatalf("captured artifacts = %#v, want [session_recap]", capturedArtifacts)
}
if !strings.Contains(stdout.String(), "narratio publish: executed=1 skipped=0 force=true; manifest=") {
t.Fatalf("stdout = %q, want publish summary", stdout.String())
}
}
func TestExecutePublishRejectsUnsupportedArgsAndFlags(t *testing.T) {
cases := []struct {
name string
args []string
want string
}{
{name: "extra positional", args: []string{"publish", "2026-05-03", "extra"}, want: "publish: unexpected positional arguments"},
{name: "force flag", args: []string{"publish", "--force"}, want: "publish: invalid flags: flag provided but not defined: -force"},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(tc.args, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), tc.want) {
t.Fatalf("stderr = %q, want %q", stderr.String(), tc.want)
}
})
}
}
func TestExecutePublishUnknownArtifactFailsValidation(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"publish", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--artifacts", "unknown_artifact"},
&stdout,
&stderr,
)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), `publish: --artifacts includes unknown artifact "unknown_artifact"`) {
t.Fatalf("stderr = %q, want unknown-artifact validation error", stderr.String())
}
}
func TestExecutePublishMissingConfigUsesRunStageLoadingPath(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"publish", "2026-05-03"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "publish: no pipeline config path provided and no default pipeline config found; searched:") {
t.Fatalf("stderr = %q, want pipeline discovery error", stderr.String())
}
}
func TestExecuteUsageIncludesAnalyzeAndPublish(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(nil, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "analyze") {
t.Fatalf("stderr = %q, want usage to include analyze", stderr.String())
}
if !strings.Contains(stderr.String(), "publish") {
t.Fatalf("stderr = %q, want usage to include publish", stderr.String())
}
}
func writeValidConfigFilesWithScriptoriumArtifacts(t *testing.T, workspaceRoot string) (string, string, string) {
t.Helper()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
f, err := os.OpenFile(pipelinePath, os.O_APPEND|os.O_WRONLY, 0)
if err != nil {
t.Fatalf("open pipeline config for append: %v", err)
@@ -136,5 +431,5 @@ scriptorium:
if _, err := f.WriteString(extra); err != nil {
t.Fatalf("append scriptorium config: %v", err)
}
return pipelinePath, sessionPath
return pipelinePath, campaignPath, sessionPath
}

View File

@@ -64,7 +64,7 @@ func TestArtifactSelectionFlagNormalize(t *testing.T) {
}
}
func TestValidateSelectedAnalyzeArtifacts(t *testing.T) {
func TestValidateSelectedArtifacts(t *testing.T) {
tests := []struct {
name string
cfg *config.Config
@@ -114,7 +114,7 @@ func TestValidateSelectedAnalyzeArtifacts(t *testing.T) {
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
err := validateSelectedAnalyzeArtifacts(tt.cfg, tt.selected)
err := validateSelectedArtifacts(tt.cfg, tt.selected)
if tt.wantErr != "" {
if err == nil {
t.Fatalf("error = nil, want %q", tt.wantErr)

View File

@@ -0,0 +1,44 @@
package app
import (
"fmt"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func resolveCampaignConfigPath(pipelineCfg *config.PipelineConfig, campaignIDFlag, campaignFileFlag string) (string, error) {
campaignID := strings.TrimSpace(campaignIDFlag)
campaignFile := strings.TrimSpace(campaignFileFlag)
if campaignID != "" && campaignFile != "" {
return "", fmt.Errorf("--campaign and --campaign-file are mutually exclusive")
}
if campaignFile != "" {
return filepath.Clean(campaignFile), nil
}
if campaignID == "" && pipelineCfg != nil {
campaignID = strings.TrimSpace(pipelineCfg.Campaigns.DefaultCampaignID)
}
if campaignID == "" {
return "", fmt.Errorf("no campaign selected; pass --campaign <id> or set pipeline.campaigns.default_campaign_id")
}
if err := validateCampaignIDToken(campaignID); err != nil {
return "", err
}
if pipelineCfg == nil || strings.TrimSpace(pipelineCfg.Campaigns.Root) == "" {
return "", fmt.Errorf("pipeline.campaigns.root is required to select campaign %q", campaignID)
}
return filepath.Clean(filepath.Join(pipelineCfg.Campaigns.Root, campaignID, "campaign.yml")), nil
}
func validateCampaignIDToken(campaignID string) error {
if filepath.IsAbs(campaignID) ||
strings.Contains(campaignID, "/") ||
strings.Contains(campaignID, `\`) ||
campaignID == "." ||
campaignID == ".." {
return fmt.Errorf("campaign id %q must be a single path segment", campaignID)
}
return nil
}

View File

@@ -0,0 +1,84 @@
package app
import (
"path/filepath"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestResolveCampaignConfigPathCampaignFileWins(t *testing.T) {
explicit := filepath.Join(t.TempDir(), "custom-campaign.yml")
got, err := resolveCampaignConfigPath(&config.PipelineConfig{}, "", explicit)
if err != nil {
t.Fatalf("resolveCampaignConfigPath() error = %v", err)
}
if got != explicit {
t.Fatalf("path = %q, want explicit path %q", got, explicit)
}
}
func TestResolveCampaignConfigPathUsesSelectedCampaignID(t *testing.T) {
dir := t.TempDir()
pipelineCfg := &config.PipelineConfig{}
pipelineCfg.Campaigns.Root = dir
got, err := resolveCampaignConfigPath(pipelineCfg, "icewind", "")
if err != nil {
t.Fatalf("resolveCampaignConfigPath() error = %v", err)
}
want := filepath.Join(dir, "icewind", "campaign.yml")
if got != filepath.Clean(want) {
t.Fatalf("path = %q, want %q", got, filepath.Clean(want))
}
}
func TestResolveCampaignConfigPathUsesDefaultCampaignID(t *testing.T) {
dir := t.TempDir()
pipelineCfg := &config.PipelineConfig{}
pipelineCfg.Campaigns.Root = dir
pipelineCfg.Campaigns.DefaultCampaignID = "dilfs"
got, err := resolveCampaignConfigPath(pipelineCfg, "", "")
if err != nil {
t.Fatalf("resolveCampaignConfigPath() error = %v", err)
}
want := filepath.Join(dir, "dilfs", "campaign.yml")
if got != filepath.Clean(want) {
t.Fatalf("path = %q, want %q", got, filepath.Clean(want))
}
}
func TestResolveCampaignConfigPathRejectsCampaignIDAndFile(t *testing.T) {
_, err := resolveCampaignConfigPath(&config.PipelineConfig{}, "dilfs", filepath.Join(t.TempDir(), "campaign.yml"))
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "mutually exclusive") {
t.Fatalf("error = %q, want mutual exclusion", err.Error())
}
}
func TestResolveCampaignConfigPathRequiresCampaignSelection(t *testing.T) {
_, err := resolveCampaignConfigPath(&config.PipelineConfig{}, "", "")
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "no campaign selected") {
t.Fatalf("error = %q, want missing selection guidance", err.Error())
}
}
func TestResolveCampaignConfigPathRejectsPathLikeCampaignID(t *testing.T) {
pipelineCfg := &config.PipelineConfig{}
pipelineCfg.Campaigns.Root = t.TempDir()
_, err := resolveCampaignConfigPath(pipelineCfg, "../icewind", "")
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "single path segment") {
t.Fatalf("error = %q, want path segment guidance", err.Error())
}
}

281
internal/app/clean.go Normal file
View File

@@ -0,0 +1,281 @@
package app
import (
"context"
"flag"
"fmt"
"io"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
// Clean removes local workspace/spool state while preserving durable cache
// state unless cache cleanup is explicitly requested.
func Clean(ctx context.Context, args []string, out io.Writer) error {
fs := flag.NewFlagSet("clean", flag.ContinueOnError)
fs.SetOutput(io.Discard)
var flags commonConfigFlags
var all bool
var dryRun bool
var clearCache bool
addCommonConfigFlags(fs, &flags)
fs.BoolVar(&all, "all", false, "clean all local session work/spool state")
fs.BoolVar(&dryRun, "dry-run", false, "print cleanup targets without deleting")
fs.BoolVar(&clearCache, "clear-cache", false, "also clear durable S3 audio cache entries")
if err := parseSessionAwareFlags("clean", fs, args, &flags.sessionID); err != nil {
return err
}
if all {
return cleanAllLocal(flags, dryRun, clearCache, out)
}
return cleanSession(ctx, flags, dryRun, clearCache, out)
}
func cleanSession(ctx context.Context, flags commonConfigFlags, dryRun, clearCache bool, out io.Writer) error {
if strings.TrimSpace(flags.sessionID) == "" {
return fmt.Errorf("clean: session_id is required unless --all is set")
}
cfg, err := loadCommandConfig(ctx, flags.pipelinePath, flags.campaignPath, flags.campaignFilePath, flags.sessionPath, flags.sessionOptions())
if err != nil {
return fmt.Errorf("clean: %w", err)
}
if cfg == nil || cfg.Pipeline == nil || cfg.Session == nil {
return fmt.Errorf("clean: resolved pipeline and session config are required")
}
campaign := strings.TrimSpace(cfg.Session.Campaign)
sessionID := strings.TrimSpace(cfg.Session.SessionID)
if campaign == "" || sessionID == "" {
return fmt.Errorf("clean: campaign and session_id are required")
}
if dryRun {
fmt.Fprintf(out, "Clean plan for %s/%s\n", campaign, sessionID)
} else {
fmt.Fprintf(out, "Cleaned %s/%s\n", campaign, sessionID)
}
workDir := artifacts.SessionWorkDirForCampaign(cfg.Pipeline.Workspace.Root, campaign, sessionID)
spoolDir := artifacts.SessionSpoolDir(cfg.Pipeline.Spool.Root, campaign, sessionID)
if err := reportCleanScopedDir(out, cfg.Pipeline.Workspace.Root, workDir, "clean.workspace.session", dryRun); err != nil {
return fmt.Errorf("clean: %w", err)
}
if err := reportCleanScopedDir(out, cfg.Pipeline.Spool.Root, spoolDir, "clean.spool.session", dryRun); err != nil {
return fmt.Errorf("clean: %w", err)
}
if clearCache {
if err := cleanSessionAudioCache(ctx, cfg, dryRun, out); err != nil {
return fmt.Errorf("clean: %w", err)
}
} else {
fmt.Fprintln(out, "Cache: preserved")
}
return nil
}
func cleanAllLocal(flags commonConfigFlags, dryRun, clearCache bool, out io.Writer) error {
if strings.TrimSpace(flags.campaignPath) != "" ||
strings.TrimSpace(flags.campaignFilePath) != "" ||
strings.TrimSpace(flags.sessionPath) != "" ||
strings.TrimSpace(flags.sessionID) != "" ||
strings.TrimSpace(flags.previousSessionID) != "" {
return fmt.Errorf("clean: --all cannot be combined with --campaign, --campaign-file, --session, a session_id, or --previous-session-id")
}
resolvedPipelinePath, err := resolvePipelineConfigPath(flags.pipelinePath)
if err != nil {
return fmt.Errorf("clean: %w", err)
}
pipelineCfg, err := config.LoadPipeline(resolvedPipelinePath)
if err != nil {
return fmt.Errorf("clean: %w", err)
}
if dryRun {
fmt.Fprintln(out, "Clean plan for all local sessions")
} else {
fmt.Fprintln(out, "Cleaned all local sessions")
}
workRoot := filepath.Join(pipelineCfg.Workspace.Root, config.PathWorkDirSegment)
if err := reportCleanScopedDir(out, pipelineCfg.Workspace.Root, workRoot, "clean.workspace.all", dryRun); err != nil {
return fmt.Errorf("clean: %w", err)
}
if err := reportCleanRootChildren(out, pipelineCfg.Spool.Root, "clean.spool.all", dryRun); err != nil {
return fmt.Errorf("clean: %w", err)
}
if clearCache {
if err := cleanAllAudioCache(pipelineCfg, dryRun, out); err != nil {
return fmt.Errorf("clean: %w", err)
}
} else {
fmt.Fprintln(out, "Cache: preserved")
}
return nil
}
func reportCleanScopedDir(out io.Writer, root, target, policy string, dryRun bool) error {
dir, err := validateScopedDir(root, target, policy)
if err != nil {
return err
}
if dryRun {
if dir.Exists {
fmt.Fprintf(out, "Would delete: %s\n", dir.TargetAbs)
} else {
fmt.Fprintf(out, "Would skip missing: %s\n", dir.TargetAbs)
}
return nil
}
if !dir.Exists {
fmt.Fprintf(out, "Missing: %s\n", dir.TargetAbs)
return nil
}
if err := os.RemoveAll(dir.TargetAbs); err != nil {
return fmt.Errorf("cleanup policy %s: remove %q: %w", policy, dir.TargetAbs, err)
}
fmt.Fprintf(out, "Deleted: %s\n", dir.TargetAbs)
return nil
}
func reportCleanRootChildren(out io.Writer, root, policy string, dryRun bool) error {
rootAbs, entries, err := cleanableRootChildren(root, policy)
if err != nil {
return err
}
if len(entries) == 0 {
if dryRun {
fmt.Fprintf(out, "Would skip empty: %s\n", rootAbs)
} else {
fmt.Fprintf(out, "Empty: %s\n", rootAbs)
}
return nil
}
for _, entry := range entries {
if dryRun {
fmt.Fprintf(out, "Would delete: %s\n", entry)
continue
}
if err := os.RemoveAll(entry); err != nil {
return fmt.Errorf("cleanup policy %s: remove %q: %w", policy, entry, err)
}
fmt.Fprintf(out, "Deleted: %s\n", entry)
}
return nil
}
func cleanableRootChildren(root, policy string) (string, []string, error) {
rootAbs, exists, err := validateCleanRoot(root, policy)
if err != nil {
return "", nil, err
}
if !exists {
return rootAbs, nil, nil
}
entries, err := os.ReadDir(rootAbs)
if err != nil {
return "", nil, fmt.Errorf("cleanup policy %s: read root %q: %w", policy, rootAbs, err)
}
out := make([]string, 0, len(entries))
for _, entry := range entries {
path := filepath.Join(rootAbs, entry.Name())
info, err := os.Lstat(path)
if err != nil {
return "", nil, fmt.Errorf("cleanup policy %s: stat child %q: %w", policy, path, err)
}
if info.Mode()&os.ModeSymlink != 0 {
return "", nil, fmt.Errorf("cleanup policy %s: refusing to delete symlink path %q", policy, path)
}
out = append(out, path)
}
return rootAbs, out, nil
}
func cleanSessionAudioCache(ctx context.Context, cfg *config.Config, dryRun bool, out io.Writer) error {
if cfg.Session.Inputs.AudioS3 == nil {
fmt.Fprintln(out, "Cache: skipped (session does not use audio_s3)")
return nil
}
if cfg.Pipeline.Storage.S3 == nil || strings.TrimSpace(cfg.Pipeline.Storage.S3.Bucket) == "" {
return fmt.Errorf("clear cache requires pipeline.storage.s3.bucket")
}
store, err := newCommandObjectStore(ctx, cfg, nil)
if err != nil {
return fmt.Errorf("initialize object store for cache cleanup: %w", err)
}
sessionPrefix := artifacts.S3SessionPrefix(cfg.Pipeline.Storage.S3.RootPrefix, cfg.Session.Campaign, cfg.Session.SessionID)
audioPrefix := artifacts.S3AudioPrefix(sessionPrefix, cfg.Session.Inputs.AudioS3.Prefix)
objects, err := store.List(ctx, audioPrefix)
if err != nil {
return fmt.Errorf("list s3 audio objects under %q: %w", audioPrefix, err)
}
count := 0
for _, obj := range objects {
key := strings.TrimSpace(obj.Key)
if key == "" || strings.HasSuffix(key, "/") || !cleanIsFlac(key) {
continue
}
cachePath, err := artifacts.S3AudioCachePath(cfg.Pipeline.Cache.Root, cfg.Pipeline.Storage.S3.Bucket, key)
if err != nil {
return err
}
deleted, err := reportCleanScopedFile(out, cfg.Pipeline.Cache.Root, cachePath, "clean.cache.session", dryRun)
if err != nil {
return err
}
if deleted {
count++
}
}
if count == 0 {
fmt.Fprintf(out, "Cache: no cached S3 audio files found for %s\n", audioPrefix)
}
return nil
}
func cleanAllAudioCache(cfg *config.PipelineConfig, dryRun bool, out io.Writer) error {
if cfg.Storage.S3 == nil || strings.TrimSpace(cfg.Storage.S3.Bucket) == "" {
return fmt.Errorf("clear cache requires pipeline.storage.s3.bucket")
}
namespaceDir, err := artifacts.S3AudioCacheNamespaceDir(cfg.Cache.Root, cfg.Storage.S3.Bucket, cfg.Storage.S3.RootPrefix)
if err != nil {
return err
}
return reportCleanScopedDir(out, cfg.Cache.Root, namespaceDir, "clean.cache.all", dryRun)
}
func reportCleanScopedFile(out io.Writer, root, target, policy string, dryRun bool) (bool, error) {
file, err := validateScopedFile(root, target, policy)
if err != nil {
return false, err
}
if dryRun {
if file.Exists {
fmt.Fprintf(out, "Would delete cache file: %s\n", file.TargetAbs)
return true, nil
}
fmt.Fprintf(out, "Would skip missing cache file: %s\n", file.TargetAbs)
return false, nil
}
if !file.Exists {
fmt.Fprintf(out, "Missing cache file: %s\n", file.TargetAbs)
return false, nil
}
if err := os.Remove(file.TargetAbs); err != nil {
return false, fmt.Errorf("cleanup policy %s: remove %q: %w", policy, file.TargetAbs, err)
}
fmt.Fprintf(out, "Deleted cache file: %s\n", file.TargetAbs)
return true, nil
}
func validateScopedFile(root, target, policy string) (scopedDir, error) {
return validateScopedTarget(root, target, policy, false)
}
func cleanIsFlac(path string) bool {
return strings.EqualFold(filepath.Ext(path), ".flac")
}

255
internal/app/clean_test.go Normal file
View File

@@ -0,0 +1,255 @@
package app
import (
"bytes"
"os"
"path/filepath"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
)
func TestExecuteCleanSessionDeletesWorkAndSpoolButPreservesCache(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
workDir := artifacts.SessionWorkDirForCampaign(workspaceRoot, "sample-campaign", "2026-05-03")
spoolDir := artifacts.SessionSpoolDir(filepath.Join(workspaceRoot, "spool"), "sample-campaign", "2026-05-03")
cachePath, err := artifacts.S3AudioCachePath(filepath.Join(workspaceRoot, "cache"), "test-bucket", "dnd/campaigns/sample-campaign/sessions/2026-05-03/audio/alice.flac")
if err != nil {
t.Fatalf("S3AudioCachePath() error = %v", err)
}
mustWriteTestFile(t, filepath.Join(workDir, "manifest.json"), "{}")
mustWriteTestFile(t, filepath.Join(spoolDir, "run-1", "audio", "alice.flac"), "audio")
mustWriteTestFile(t, cachePath, "cached-audio")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
cleanAssertMissing(t, workDir)
cleanAssertMissing(t, spoolDir)
cleanAssertExists(t, cachePath)
if !strings.Contains(stdout.String(), "Cache: preserved") {
t.Fatalf("stdout = %q, want cache preserved", stdout.String())
}
}
func TestExecuteCleanSessionDryRunDeletesNothing(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
workDir := artifacts.SessionWorkDirForCampaign(workspaceRoot, "sample-campaign", "2026-05-03")
spoolDir := artifacts.SessionSpoolDir(filepath.Join(workspaceRoot, "spool"), "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workDir, "manifest.json"), "{}")
mustWriteTestFile(t, filepath.Join(spoolDir, "run-1", "audio", "alice.flac"), "audio")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--dry-run"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
cleanAssertExists(t, workDir)
cleanAssertExists(t, spoolDir)
if !strings.Contains(stdout.String(), "Would delete:") {
t.Fatalf("stdout = %q, want dry-run delete plan", stdout.String())
}
}
func TestExecuteCleanMissingSessionPathsSucceeds(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "Missing:") {
t.Fatalf("stdout = %q, want missing path output", stdout.String())
}
}
func TestExecuteCleanSessionClearCacheRemovesOnlyS3AudioCache(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
if err := os.WriteFile(sessionPath, []byte(`session_id: 2026-05-03
inputs:
audio_s3:
prefix: audio/
`), 0o644); err != nil {
t.Fatalf("write session: %v", err)
}
audioKey := "dnd/campaigns/sample-campaign/sessions/2026-05-03/audio/alice.flac"
fake := &storage.FakeBackend{}
fake.SeedObject(storage.FakeObject{Key: audioKey, Data: []byte("audio")})
var storeInitCalls int
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{sessionPath})
cacheRoot := filepath.Join(workspaceRoot, "cache")
cachePath, err := artifacts.S3AudioCachePath(cacheRoot, "test-bucket", audioKey)
if err != nil {
t.Fatalf("S3AudioCachePath() error = %v", err)
}
otherCachePath, err := artifacts.S3AudioCachePath(cacheRoot, "test-bucket", "dnd/campaigns/other/sessions/2026-05-03/audio/bob.flac")
if err != nil {
t.Fatalf("S3AudioCachePath() error = %v", err)
}
mustWriteTestFile(t, cachePath, "cached-audio")
mustWriteTestFile(t, otherCachePath, "other-audio")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--clear-cache"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
cleanAssertMissing(t, cachePath)
cleanAssertExists(t, otherCachePath)
if storeInitCalls != 1 {
t.Fatalf("object store init calls = %d, want 1", storeInitCalls)
}
}
func TestExecuteCleanLocalAudioClearCacheIsNoop(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--clear-cache"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "Cache: skipped (session does not use audio_s3)") {
t.Fatalf("stdout = %q, want local audio cache no-op", stdout.String())
}
}
func TestExecuteCleanAllDeletesWorkAndSpoolContentsButPreservesCache(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, _, _ := writeValidConfigFiles(t, workspaceRoot)
workRoot := filepath.Join(workspaceRoot, "work")
spoolRoot := filepath.Join(workspaceRoot, "spool")
cachePath := filepath.Join(workspaceRoot, "cache", "keep.txt")
mustWriteTestFile(t, filepath.Join(workRoot, "sample-campaign", "2026-05-03", "manifest.json"), "{}")
mustWriteTestFile(t, filepath.Join(spoolRoot, "sample-campaign", "2026-05-03", "run-1", "audio", "alice.flac"), "audio")
mustWriteTestFile(t, cachePath, "cache")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "--config", pipelinePath, "--all"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
cleanAssertMissing(t, workRoot)
cleanAssertExists(t, spoolRoot)
cleanAssertMissing(t, filepath.Join(spoolRoot, "sample-campaign"))
cleanAssertExists(t, cachePath)
}
func TestExecuteCleanAllClearCacheRemovesS3AudioNamespaceOnly(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, _, _ := writeValidConfigFiles(t, workspaceRoot)
cacheRoot := filepath.Join(workspaceRoot, "cache")
audioCachePath, err := artifacts.S3AudioCachePath(cacheRoot, "test-bucket", "dnd/campaigns/sample-campaign/sessions/2026-05-03/audio/alice.flac")
if err != nil {
t.Fatalf("S3AudioCachePath() error = %v", err)
}
otherCachePath, err := artifacts.S3AudioCachePath(cacheRoot, "test-bucket", "other-root/campaigns/sample-campaign/sessions/2026-05-03/audio/alice.flac")
if err != nil {
t.Fatalf("S3AudioCachePath() error = %v", err)
}
mustWriteTestFile(t, audioCachePath, "cached-audio")
mustWriteTestFile(t, otherCachePath, "other-cache")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "--config", pipelinePath, "--all", "--clear-cache"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
cleanAssertMissing(t, audioCachePath)
cleanAssertExists(t, otherCachePath)
}
func TestExecuteCleanAllRejectsSessionScopedFlags(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, _ := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--all"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "--all cannot be combined") {
t.Fatalf("stderr = %q, want --all conflict", stderr.String())
}
}
func TestCleanRequiresSessionID(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clean"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "session_id is required unless --all is set") {
t.Fatalf("stderr = %q, want missing session-id", stderr.String())
}
}
func TestCleanRejectsUnsafeTargets(t *testing.T) {
root := t.TempDir()
outside := t.TempDir()
if err := reportCleanScopedDir(&bytes.Buffer{}, root, filepath.Join(outside, "target"), "test.outside", false); err == nil {
t.Fatal("outside target error = nil, want error")
}
if err := reportCleanScopedDir(&bytes.Buffer{}, root, root, "test.root", false); err == nil {
t.Fatal("root target error = nil, want error")
}
filePath := filepath.Join(root, "file.txt")
mustWriteTestFile(t, filePath, "file")
if err := reportCleanScopedDir(&bytes.Buffer{}, root, filePath, "test.file", false); err == nil {
t.Fatal("file target error = nil, want error")
}
symlinkPath := filepath.Join(root, "link")
if err := os.Symlink(filepath.Join(root, "missing"), symlinkPath); err != nil {
t.Fatalf("Symlink() error = %v", err)
}
if err := reportCleanScopedDir(&bytes.Buffer{}, root, symlinkPath, "test.symlink", false); err == nil {
t.Fatal("symlink target error = nil, want error")
}
}
func TestClearIsNotCommandAlias(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"clear"}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), `unknown command: "clear"`) {
t.Fatalf("stderr = %q, want unknown clear command", stderr.String())
}
}
func cleanAssertExists(t *testing.T, path string) {
t.Helper()
if _, err := os.Stat(path); err != nil {
t.Fatalf("expected %q to exist: %v", path, err)
}
}
func cleanAssertMissing(t *testing.T, path string) {
t.Helper()
if _, err := os.Stat(path); !os.IsNotExist(err) {
t.Fatalf("expected %q to be missing, stat err=%v", path, err)
}
}

View File

@@ -0,0 +1,82 @@
package app
import (
"fmt"
"os"
"path/filepath"
"strings"
)
func validateScopedTarget(root, target, policy string, requireDir bool) (scopedDir, error) {
cleanRoot := strings.TrimSpace(root)
cleanTarget := strings.TrimSpace(target)
if cleanRoot == "" {
return scopedDir{}, fmt.Errorf("cleanup policy %s: root path is required", policy)
}
if cleanTarget == "" {
return scopedDir{}, fmt.Errorf("cleanup policy %s: target path is required", policy)
}
rootAbs, err := filepath.Abs(cleanRoot)
if err != nil {
return scopedDir{}, fmt.Errorf("cleanup policy %s: resolve root %q: %w", policy, cleanRoot, err)
}
targetAbs, err := filepath.Abs(cleanTarget)
if err != nil {
return scopedDir{}, fmt.Errorf("cleanup policy %s: resolve target %q: %w", policy, cleanTarget, err)
}
rel, err := filepath.Rel(rootAbs, targetAbs)
if err != nil {
return scopedDir{}, fmt.Errorf("cleanup policy %s: relative path from %q to %q: %w", policy, rootAbs, targetAbs, err)
}
if rel == "." {
return scopedDir{}, fmt.Errorf("cleanup policy %s: refusing to delete root directory %q", policy, rootAbs)
}
if rel == ".." || strings.HasPrefix(rel, ".."+string(filepath.Separator)) {
return scopedDir{}, fmt.Errorf("cleanup policy %s: refusing to delete path outside root: root=%q target=%q", policy, rootAbs, targetAbs)
}
info, err := os.Lstat(targetAbs)
if err != nil {
if os.IsNotExist(err) {
return scopedDir{RootAbs: rootAbs, TargetAbs: targetAbs, Exists: false}, nil
}
return scopedDir{}, fmt.Errorf("cleanup policy %s: stat target %q: %w", policy, targetAbs, err)
}
if info.Mode()&os.ModeSymlink != 0 {
return scopedDir{}, fmt.Errorf("cleanup policy %s: refusing to delete symlink path %q", policy, targetAbs)
}
if requireDir && !info.IsDir() {
return scopedDir{}, fmt.Errorf("cleanup policy %s: target %q is not a directory", policy, targetAbs)
}
if !requireDir && info.IsDir() {
return scopedDir{}, fmt.Errorf("cleanup policy %s: target %q is a directory", policy, targetAbs)
}
return scopedDir{RootAbs: rootAbs, TargetAbs: targetAbs, Exists: true}, nil
}
func validateCleanRoot(root, policy string) (string, bool, error) {
cleanRoot := strings.TrimSpace(root)
if cleanRoot == "" {
return "", false, fmt.Errorf("cleanup policy %s: root path is required", policy)
}
rootAbs, err := filepath.Abs(cleanRoot)
if err != nil {
return "", false, fmt.Errorf("cleanup policy %s: resolve root %q: %w", policy, cleanRoot, err)
}
info, err := os.Lstat(rootAbs)
if err != nil {
if os.IsNotExist(err) {
return rootAbs, false, nil
}
return "", false, fmt.Errorf("cleanup policy %s: stat root %q: %w", policy, rootAbs, err)
}
if info.Mode()&os.ModeSymlink != 0 {
return "", false, fmt.Errorf("cleanup policy %s: refusing to clean symlink root %q", policy, rootAbs)
}
if !info.IsDir() {
return "", false, fmt.Errorf("cleanup policy %s: root %q is not a directory", policy, rootAbs)
}
return rootAbs, true, nil
}

View File

@@ -0,0 +1,103 @@
package app
import (
"os"
"path/filepath"
"strings"
"testing"
)
func TestCleanValidateScopedDirAndFile(t *testing.T) {
root := t.TempDir()
dirTarget := filepath.Join(root, "runs", "run-1")
fileTarget := filepath.Join(root, "cache", "a.flac")
if err := os.MkdirAll(dirTarget, 0o755); err != nil {
t.Fatalf("MkdirAll(dirTarget) error = %v", err)
}
if err := os.MkdirAll(filepath.Dir(fileTarget), 0o755); err != nil {
t.Fatalf("MkdirAll(file parent) error = %v", err)
}
if err := os.WriteFile(fileTarget, []byte("audio"), 0o644); err != nil {
t.Fatalf("WriteFile(fileTarget) error = %v", err)
}
if _, err := validateScopedDir(root, dirTarget, "test.dir"); err != nil {
t.Fatalf("validateScopedDir() error = %v", err)
}
if _, err := validateScopedFile(root, fileTarget, "test.file"); err != nil {
t.Fatalf("validateScopedFile() error = %v", err)
}
}
func TestCleanValidateScopedTargetSafetyRules(t *testing.T) {
root := t.TempDir()
outside := t.TempDir()
target := filepath.Join(root, "runs", "run-1")
if err := os.MkdirAll(target, 0o755); err != nil {
t.Fatalf("MkdirAll(target) error = %v", err)
}
fileTarget := filepath.Join(root, "cache", "a.flac")
if err := os.MkdirAll(filepath.Dir(fileTarget), 0o755); err != nil {
t.Fatalf("MkdirAll(file parent) error = %v", err)
}
if err := os.WriteFile(fileTarget, []byte("audio"), 0o644); err != nil {
t.Fatalf("WriteFile(fileTarget) error = %v", err)
}
symlinkTarget := filepath.Join(root, "symlink")
if err := os.Symlink(target, symlinkTarget); err != nil {
t.Fatalf("Symlink() error = %v", err)
}
if _, err := validateScopedDir(root, root, "test.root"); err == nil || !strings.Contains(err.Error(), "refusing to delete root directory") {
t.Fatalf("validateScopedDir(root) error = %v, want root deletion rejection", err)
}
if _, err := validateScopedDir(root, filepath.Join(outside, "x"), "test.outside"); err == nil || !strings.Contains(err.Error(), "outside root") {
t.Fatalf("validateScopedDir(outside) error = %v, want outside-root rejection", err)
}
if _, err := validateScopedDir(root, fileTarget, "test.file-as-dir"); err == nil || !strings.Contains(err.Error(), "is not a directory") {
t.Fatalf("validateScopedDir(file) error = %v, want not-a-directory rejection", err)
}
if _, err := validateScopedFile(root, target, "test.dir-as-file"); err == nil || !strings.Contains(err.Error(), "is a directory") {
t.Fatalf("validateScopedFile(dir) error = %v, want is-a-directory rejection", err)
}
if _, err := validateScopedDir(root, symlinkTarget, "test.symlink"); err == nil || !strings.Contains(err.Error(), "refusing to delete symlink path") {
t.Fatalf("validateScopedDir(symlink) error = %v, want symlink rejection", err)
}
}
func TestCleanableRootChildrenRejectsSymlinkChild(t *testing.T) {
root := t.TempDir()
realChild := filepath.Join(root, "runs")
if err := os.MkdirAll(realChild, 0o755); err != nil {
t.Fatalf("MkdirAll(realChild) error = %v", err)
}
if err := os.Symlink(realChild, filepath.Join(root, "link")); err != nil {
t.Fatalf("Symlink() error = %v", err)
}
_, _, err := cleanableRootChildren(root, "test.root.children")
if err == nil || !strings.Contains(err.Error(), "refusing to delete symlink path") {
t.Fatalf("cleanableRootChildren() error = %v, want symlink rejection", err)
}
}
func TestCleanValidateScopedTargetMissing(t *testing.T) {
root := t.TempDir()
missingDir := filepath.Join(root, "runs", "missing")
got, err := validateScopedDir(root, missingDir, "test.missing")
if err != nil {
t.Fatalf("validateScopedDir(missing) error = %v", err)
}
if got.Exists {
t.Fatalf("validateScopedDir(missing).Exists = true, want false")
}
missingFile := filepath.Join(root, "cache", "missing.flac")
got, err = validateScopedFile(root, missingFile, "test.missing.file")
if err != nil {
t.Fatalf("validateScopedFile(missing) error = %v", err)
}
if got.Exists {
t.Fatalf("validateScopedFile(missing).Exists = true, want false")
}
}

View File

@@ -7,7 +7,7 @@ import (
"strings"
)
var supportedCommands = []string{"run", "plan", "status", "resume", "run-stage", "restore"}
var supportedCommands = []string{"run", "run-stage", "analyze", "publish", "clean", "session"}
// Execute dispatches CLI commands and returns a process exit code.
func Execute(args []string, stdout, stderr io.Writer) int {
@@ -24,16 +24,16 @@ func Execute(args []string, stdout, stderr io.Writer) int {
switch cmd {
case "run":
err = Run(ctx, cmdArgs, stdout)
case "plan":
err = Plan(ctx, cmdArgs, stdout)
case "status":
err = Status(ctx, cmdArgs, stdout)
case "resume":
err = Resume(ctx, cmdArgs, stdout)
case "run-stage":
err = RunStage(ctx, cmdArgs, stdout)
case "restore":
err = Restore(ctx, cmdArgs, stdout)
case "analyze":
err = Analyze(ctx, cmdArgs, stdout)
case "publish":
err = Publish(ctx, cmdArgs, stdout)
case "session":
err = Session(ctx, cmdArgs, stdout)
case "clean":
err = Clean(ctx, cmdArgs, stdout)
default:
fmt.Fprintf(stderr, "unknown command: %q\n\n", cmd)
printUsage(stderr)
@@ -51,8 +51,3 @@ func Execute(args []string, stdout, stderr io.Writer) int {
func printUsage(w io.Writer) {
fmt.Fprintf(w, "Usage: narratio <%s>\n", strings.Join(supportedCommands, "|"))
}
func placeholder(out io.Writer, command string) error {
_, err := fmt.Fprintf(out, "narratio %s: not yet implemented\n", command)
return err
}

View File

@@ -24,19 +24,17 @@ func TestExecuteValidCommands(t *testing.T) {
}))
defer srv.Close()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL)
manifestPath := writeManifestPathForExecute(t)
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL)
cases := []struct {
name string
args []string
wantOut string
}{
{name: "run", args: []string{"run", "--config", pipelinePath, "--session", sessionPath}, wantOut: "narratio run: session 2026-05-03; executed=9 skipped=0; manifest="},
{name: "plan", args: []string{"plan", "--config", pipelinePath, "--session", sessionPath}, wantOut: "prepare: skip\ntranscribe: skip\nmerge: skip\npolish: skip\nnormalize: skip\ntrim: skip\nanalyze: skip\narchive: skip\nnotify: skip"},
{name: "status", args: []string{"status", "--manifest", manifestPath}, wantOut: "session_id: 2026-05-03"},
{name: "resume", args: []string{"resume", "--config", pipelinePath, "--session", sessionPath}, wantOut: "narratio resume: session 2026-05-03 has no remaining stages"},
{name: "run-stage", args: []string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "polish"}, wantOut: "narratio run-stage: stage=polish executed=0 skipped=1 force=false; manifest="},
{name: "run", args: []string{"run", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "narratio run: session 2026-05-03; executed=11 skipped=1; manifest="},
{name: "session plan", args: []string{"session", "plan", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "prepare: skip\ntranscribe: skip\nmerge: skip\npolish: skip\nnormalize: skip\ntrim: skip\nextract: run\nrender: skip\nanalyze: skip\npublish: skip\nnotify: skip"},
{name: "session status", args: []string{"session", "status", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "Session: 2026-05-03"},
{name: "run-stage", args: []string{"run-stage", "polish", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "narratio run-stage: stage=polish executed=0 skipped=1 force=false; manifest="},
}
for _, tc := range cases {
@@ -64,13 +62,13 @@ func TestExecuteMissingRequiredFlags(t *testing.T) {
args []string
want string
}{
{name: "run missing flags", args: []string{"run"}, want: "run: no pipeline config path provided and no default pipeline config found; searched:"},
{name: "plan missing flags", args: []string{"plan"}, want: "plan: no pipeline config path provided and no default pipeline config found; searched:"},
{name: "status missing flags", args: []string{"status"}, want: "status: --manifest is required"},
{name: "resume missing flags", args: []string{"resume"}, want: "resume: no pipeline config path provided and no default pipeline config found; searched:"},
{name: "run-stage missing name", args: []string{"run-stage", "--config", "a", "--session", "b"}, want: "run-stage: expected exactly one stage name"},
{name: "run-stage missing config flags", args: []string{"run-stage", "polish"}, want: "run-stage: no pipeline config path provided and no default pipeline config found; searched:"},
{name: "run missing config uses defaults", args: []string{"run", "--session", "session.yml"}, want: "run: no pipeline config path provided and no default pipeline config found; searched:"},
{name: "run missing session", args: []string{"run"}, want: "run: session_id is required"},
{name: "plan old top-level removed", args: []string{"plan"}, want: `unknown command: "plan"`},
{name: "status old top-level removed", args: []string{"status"}, want: `unknown command: "status"`},
{name: "resume removed", args: []string{"resume"}, want: `unknown command: "resume"`},
{name: "run-stage missing name", args: []string{"run-stage", "--config", "a", "--session", "b"}, want: "run-stage: expected stage name and session_id"},
{name: "run-stage missing session", args: []string{"run-stage", "polish"}, want: "run-stage: expected stage name and session_id"},
{name: "run missing config uses defaults", args: []string{"run", "2026-05-03", "--session", "session.yml"}, want: "run: no pipeline config path provided and no default pipeline config found; searched:"},
}
for _, tc := range cases {
@@ -94,12 +92,12 @@ func TestExecuteMissingRequiredFlags(t *testing.T) {
func TestExecuteRunStageUnknownFails(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, "https://example.com/transcribe")
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot, "https://example.com/transcribe")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "unknown"}, &stdout, &stderr)
code := Execute([]string{"run-stage", "unknown", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
@@ -108,16 +106,32 @@ func TestExecuteRunStageUnknownFails(t *testing.T) {
}
}
func TestExecuteRunStageNormalizeIsAccepted(t *testing.T) {
func TestExecuteRunStageArchiveAliasFails(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, "https://example.com/transcribe")
workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "processed.json"), `{"segments":[{"id":1}]}`)
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot, "https://example.com/transcribe")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "normalize"}, &stdout, &stderr)
code := Execute([]string{"run-stage", "archive", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), `unknown stage "archive"`) {
t.Fatalf("stderr = %q, want unknown stage alias error", stderr.String())
}
}
func TestExecuteRunStageNormalizeIsAccepted(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot, "https://example.com/transcribe")
workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "polished.json"), `{"segments":[{"id":1}]}`)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"run-stage", "normalize", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
@@ -136,19 +150,19 @@ func TestExecuteRunStageTranscribeUsesConfiguredWhisperXServer(t *testing.T) {
}))
defer srv.Close()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL)
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "prepare"}, &stdout, &stderr)
code := Execute([]string{"run-stage", "prepare", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("prepare exit code = %d, want 0; stderr=%q", code, stderr.String())
}
stdout.Reset()
stderr.Reset()
code = Execute([]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "--force", "transcribe"}, &stdout, &stderr)
code = Execute([]string{"run-stage", "transcribe", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--force"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("transcribe exit code = %d, want 0; stderr=%q", code, stderr.String())
}
@@ -188,6 +202,7 @@ func TestExecuteRunStagePolishLoadsCredentialFromSecretsDir(t *testing.T) {
t.Setenv("GO_WANT_APP_AUDITA_HELPER", "1")
pipelinePath := filepath.Join(configDir, "pipeline.yml")
campaignPath := writeAppTestCampaignConfig(t, configDir)
sessionPath := filepath.Join(configDir, "session.yml")
pipelineYAML := `workspace:
root: ` + workspaceRoot + `
@@ -202,8 +217,6 @@ seriatim:
audita:
binary: ` + auditaBinary + `
llm_api_key_env: OPENROUTER_API_KEY
analyzer:
timeout: 20m
notification:
timeout: 10s
`
@@ -214,6 +227,8 @@ inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
players_file: ./players.yml
party_file: ./party.yml
`
if err := os.WriteFile(pipelinePath, []byte(pipelineYAML), 0o644); err != nil {
t.Fatalf("write pipeline.yml: %v", err)
@@ -234,12 +249,12 @@ inputs:
})
workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", sessionID)
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "merged.json"), `{"schema":"seriatim-intermediate","segments":[]}`)
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "base.json"), `{"schema":"seriatim-intermediate","segments":[]}`)
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "[]\n")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "--force", "polish"}, &stdout, &stderr)
code := Execute([]string{"run-stage", "polish", sessionID, "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath, "--force"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
@@ -252,6 +267,7 @@ func TestExecuteRunFailsWhenConfiguredSecretsDirMissing(t *testing.T) {
workspaceRoot := t.TempDir()
configDir := t.TempDir()
pipelinePath := filepath.Join(configDir, "pipeline.yml")
campaignPath := writeAppTestCampaignConfig(t, configDir)
sessionPath := filepath.Join(configDir, "session.yml")
pipelineYAML := `workspace:
@@ -266,8 +282,6 @@ seriatim:
binary: seriatim
audita:
binary: audita
analyzer:
timeout: 20m
notification:
timeout: 10s
`
@@ -278,6 +292,8 @@ inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
players_file: ./players.yml
party_file: ./party.yml
`
if err := os.WriteFile(pipelinePath, []byte(pipelineYAML), 0o644); err != nil {
t.Fatalf("write pipeline.yml: %v", err)
@@ -288,7 +304,7 @@ inputs:
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"run", "--config", pipelinePath, "--session", sessionPath}, &stdout, &stderr)
code := Execute([]string{"run", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
@@ -305,24 +321,109 @@ func TestExecuteUsesDefaultPipelineConfigPathWhenConfigFlagOmitted(t *testing.T)
}))
defer srv.Close()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL)
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL)
originalDefaults := append([]string(nil), config.DefaultPipelineConfigSearchPaths...)
config.DefaultPipelineConfigSearchPaths = []string{pipelinePath}
defer func() {
config.DefaultPipelineConfigSearchPaths = originalDefaults
}()
_ = campaignPath
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"run", "--session", sessionPath}, &stdout, &stderr)
code := Execute([]string{"run", "2026-05-03", "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "narratio run: session 2026-05-03; executed=9 skipped=0; manifest=") {
if !strings.Contains(stdout.String(), "narratio run: session 2026-05-03; executed=11 skipped=1; manifest=") {
t.Fatalf("stdout = %q, want successful run output", stdout.String())
}
}
func TestExecuteMissingCampaignConfigReportsRegistryPath(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
if err := os.Remove(campaignPath); err != nil {
t.Fatalf("remove campaign config: %v", err)
}
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"run", "2026-05-03", "--config", pipelinePath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if stdout.Len() != 0 {
t.Fatalf("stdout = %q, want empty", stdout.String())
}
if !strings.Contains(stderr.String(), "load campaign config") {
t.Fatalf("stderr = %q, want campaign discovery failure", stderr.String())
}
if !strings.Contains(stderr.String(), filepath.ToSlash(filepath.Join("campaigns", "sample-campaign", "campaign.yml"))) {
t.Fatalf("stderr = %q, want campaign registry path", stderr.String())
}
}
func TestExecuteUsesPipelineDefaultCampaignID(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, _, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "status", "2026-05-03", "--config", pipelinePath, "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "Campaign: sample-campaign") {
t.Fatalf("stdout = %q, want default campaign", stdout.String())
}
}
func TestExecuteCampaignIDSelectsRegistryCampaign(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
campaignRoot := filepath.Dir(filepath.Dir(campaignPath))
otherDir := filepath.Join(campaignRoot, "icewind")
mustWriteTestFile(t, filepath.Join(otherDir, "campaign.yml"), `campaign_id: icewind
inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
players_file: ./players.yml
party_file: ./party.yml
`)
mustWriteTestFile(t, filepath.Join(otherDir, "speakers.yml"), "match:\n - speaker: Alice\n match: [\"alice\"]\n")
mustWriteTestFile(t, filepath.Join(otherDir, "autocorrect.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(otherDir, "glossary.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(otherDir, "players.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(otherDir, "party.yml"), "[]\n")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "status", "2026-05-03", "--config", pipelinePath, "--campaign", "icewind", "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "Campaign: icewind") {
t.Fatalf("stdout = %q, want selected campaign", stdout.String())
}
}
func TestExecuteRejectsCampaignIDAndCampaignFile(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"session", "status", "2026-05-03", "--config", pipelinePath, "--campaign", "sample-campaign", "--campaign-file", campaignPath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "mutually exclusive") {
t.Fatalf("stderr = %q, want mutually exclusive error", stderr.String())
}
}
func TestExecuteInvalidCommand(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
@@ -359,29 +460,42 @@ func TestExecuteMissingCommand(t *testing.T) {
}
}
func writeValidConfigFiles(t *testing.T, workspaceRoot string, transcribeURL ...string) (string, string) {
func writeValidConfigFiles(t *testing.T, workspaceRoot string, transcribeURL ...string) (string, string, string) {
t.Helper()
dir := t.TempDir()
pipelinePath := filepath.Join(dir, "pipeline.yml")
campaignRoot := filepath.Join(dir, "campaigns")
campaignDir := filepath.Join(campaignRoot, "sample-campaign")
campaignPath := filepath.Join(campaignDir, "campaign.yml")
sessionPath := filepath.Join(dir, "session.yml")
url := "https://example.com/transcribe"
if len(transcribeURL) > 0 && strings.TrimSpace(transcribeURL[0]) != "" {
url = transcribeURL[0]
}
seriatimBinary := writeSeriatimAppTestWrapper(t)
scriptoriumBinary := writeScriptoriumAppTestWrapper(t)
auditaBinary := writeAuditaAppTestWrapper(t)
t.Setenv("GO_WANT_APP_SERIATIM_HELPER", "1")
t.Setenv("GO_WANT_APP_SCRIPTORIUM_HELPER", "1")
t.Setenv("GO_WANT_APP_AUDITA_HELPER", "1")
t.Setenv("AUDITA_LLM_API_KEY", "test-audita-key")
t.Setenv("PATH", filepath.Dir(scriptoriumBinary)+string(os.PathListSeparator)+os.Getenv("PATH"))
pipelineYAML := `workspace:
root: ` + workspaceRoot + `
campaigns:
root: ` + campaignRoot + `
default_campaign_id: sample-campaign
cache:
root: ` + filepath.Join(workspaceRoot, "cache") + `
spool:
root: ` + filepath.Join(workspaceRoot, "spool") + `
storage:
backend: s3
s3:
bucket: test-bucket
archive:
publish:
enabled: true
upload_run: false
whisperx:
@@ -398,36 +512,61 @@ seriatim:
report: true
audita:
binary: ` + auditaBinary + `
analyzer:
timeout: 20m
artifacts:
output_dir: artifacts
notification:
timeout: 10s
`
sessionYAML := `session_id: 2026-05-03
campaign: sample-campaign
inputs:
audio_dir: ./audio
`
campaignYAML := `campaign_id: sample-campaign
inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
players_file: ./players.yml
party_file: ./party.yml
`
if err := os.WriteFile(pipelinePath, []byte(pipelineYAML), 0o644); err != nil {
t.Fatalf("write pipeline config: %v", err)
}
if err := os.MkdirAll(campaignDir, 0o755); err != nil {
t.Fatalf("create campaign dir: %v", err)
}
if err := os.WriteFile(campaignPath, []byte(campaignYAML), 0o644); err != nil {
t.Fatalf("write campaign config: %v", err)
}
if err := os.WriteFile(sessionPath, []byte(sessionYAML), 0o644); err != nil {
t.Fatalf("write session config: %v", err)
}
mustWriteTestFile(t, filepath.Join(dir, "speakers.yml"), "match:\n - speaker: Alice\n match: [\"alice\"]\n")
mustWriteTestFile(t, filepath.Join(dir, "autocorrect.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(dir, "glossary.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(campaignDir, "speakers.yml"), "match:\n - speaker: Alice\n match: [\"alice\"]\n")
mustWriteTestFile(t, filepath.Join(campaignDir, "autocorrect.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(campaignDir, "glossary.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(campaignDir, "players.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(campaignDir, "party.yml"), "[]\n")
mustWriteTestFile(t, filepath.Join(dir, "audio", "alice.flac"), "audio-bytes")
return pipelinePath, sessionPath
return pipelinePath, campaignPath, sessionPath
}
func writeAppTestCampaignConfig(t *testing.T, dir string) string {
t.Helper()
campaignPath := filepath.Join(dir, "campaign.yml")
campaignYAML := `campaign_id: sample-campaign
inputs:
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
players_file: ./players.yml
party_file: ./party.yml
`
if err := os.WriteFile(campaignPath, []byte(campaignYAML), 0o644); err != nil {
t.Fatalf("write campaign.yml: %v", err)
}
return campaignPath
}
func writeManifestPathForExecute(t *testing.T) string {
@@ -469,6 +608,60 @@ func writeSeriatimAppTestWrapper(t *testing.T) string {
return path
}
func writeScriptoriumAppTestWrapper(t *testing.T) string {
t.Helper()
exe, err := os.Executable()
if err != nil {
t.Fatalf("os.Executable() error = %v", err)
}
path := filepath.Join(t.TempDir(), "scriptorium")
content := "#!/bin/sh\nexec \"" + exe + "\" -test.run=TestScriptoriumAppHelper -- \"$@\"\n"
if err := os.WriteFile(path, []byte(content), 0o755); err != nil {
t.Fatalf("WriteFile(%q): %v", path, err)
}
return path
}
func TestScriptoriumAppHelper(t *testing.T) {
if os.Getenv("GO_WANT_APP_SCRIPTORIUM_HELPER") != "1" {
return
}
args := os.Args
start := -1
for i := range args {
if args[i] == "--" {
start = i + 1
break
}
}
if start < 0 || start >= len(args) {
_, _ = os.Stderr.WriteString("missing -- args separator\n")
os.Exit(2)
}
runArgs := args[start:]
outputPath := appSeriatimFlagValue(runArgs, "--out")
if strings.TrimSpace(outputPath) == "" {
outputPath = appSeriatimFlagValue(runArgs, "--output")
}
if strings.TrimSpace(outputPath) == "" {
_, _ = os.Stderr.WriteString("missing output flag\n")
os.Exit(2)
}
if err := os.MkdirAll(filepath.Dir(outputPath), 0o755); err != nil {
_, _ = os.Stderr.WriteString(fmt.Sprintf("mkdir output dir: %v\n", err))
os.Exit(2)
}
if err := os.WriteFile(outputPath, []byte(`{"trim_action":"copy","warnings":[]}`), 0o644); err != nil {
_, _ = os.Stderr.WriteString(fmt.Sprintf("write output: %v\n", err))
os.Exit(2)
}
_, _ = os.Stdout.WriteString("scriptorium helper stdout\n")
_, _ = os.Stderr.WriteString("scriptorium helper stderr\n")
os.Exit(0)
}
func TestSeriatimAppHelper(t *testing.T) {
if os.Getenv("GO_WANT_APP_SERIATIM_HELPER") != "1" {
return

View File

@@ -0,0 +1,141 @@
package app
import (
"context"
"fmt"
"os"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
type pipelineCampaignConfig struct {
PipelinePath string
CampaignPath string
Pipeline *config.PipelineConfig
Campaign *config.CampaignConfig
}
func loadCommandConfig(ctx context.Context, pipelineFlag, campaignFlag, campaignFileFlag, sessionFlag string, sessionOpts config.SessionLoadOptions) (*config.Config, error) {
base, err := loadPipelineCampaignConfig(pipelineFlag, campaignFlag, campaignFileFlag)
if err != nil {
return nil, err
}
if explicitSession := strings.TrimSpace(sessionFlag); explicitSession != "" {
return config.LoadWithSessionOptions(base.PipelinePath, base.CampaignPath, explicitSession, sessionOpts)
}
discoveredSession, err := discoverSessionConfigPathWithCandidates(config.DefaultSessionConfigSearchPaths)
if err != nil {
return nil, err
}
if discoveredSession.Path != "" {
return config.LoadWithSessionOptions(base.PipelinePath, base.CampaignPath, discoveredSession.Path, sessionOpts)
}
sessionID := strings.TrimSpace(sessionOpts.SessionID)
if sessionID == "" {
return nil, missingSessionConfigError(discoveredSession.Searched, "remote session loading requires a session_id")
}
sessionPrefix := artifacts.S3SessionPrefix(base.Pipeline.Storage.S3.RootPrefix, config.CampaignID(base.Campaign), sessionID)
remoteKey := artifacts.S3SessionConfigKey(sessionPrefix)
partialCfg := &config.Config{
Pipeline: base.Pipeline,
Campaign: base.Campaign,
PipelinePath: base.PipelinePath,
CampaignPath: base.CampaignPath,
}
store, err := newCommandObjectStore(ctx, partialCfg, nil)
if err != nil {
return nil, missingSessionConfigError(discoveredSession.Searched, fmt.Sprintf("remote session %q unavailable: %v", remoteKey, err))
}
sessionInfo, err := findRemoteSessionConfig(ctx, store, sessionPrefix, remoteKey)
if err != nil {
return nil, missingSessionConfigError(discoveredSession.Searched, err.Error())
}
sessionTempPath, err := storage.DownloadObjectToTemp(ctx, store, remoteKey, "narratio-session-*.yml")
if err != nil {
return nil, missingSessionConfigError(discoveredSession.Searched, fmt.Sprintf("remote session %q download failed: %v", remoteKey, err))
}
sessionBytes, err := os.ReadFile(sessionTempPath)
if err != nil {
return nil, fmt.Errorf("read downloaded remote session %q: %w", sessionTempPath, err)
}
sessionCfg, err := config.LoadSessionBytesWithOptions("s3://"+s3BucketName(base.Pipeline)+"/"+remoteKey, sessionBytes, sessionOpts)
if err != nil {
return nil, err
}
return config.Resolve(
base.PipelinePath,
base.Pipeline,
base.CampaignPath,
base.Campaign,
sessionTempPath,
sessionCfg,
config.SessionSource{
Source: "session_config.s3",
LocalPath: sessionTempPath,
S3Bucket: s3BucketName(base.Pipeline),
S3Key: remoteKey,
S3Size: sessionInfo.Size,
S3ETag: sessionInfo.ETag,
SpoolPath: sessionTempPath,
},
)
}
func loadPipelineCampaignConfig(pipelineFlag, campaignFlag, campaignFileFlag string) (*pipelineCampaignConfig, error) {
resolvedPipelinePath, err := resolvePipelineConfigPath(pipelineFlag)
if err != nil {
return nil, err
}
pipelineCfg, err := config.LoadPipeline(resolvedPipelinePath)
if err != nil {
return nil, err
}
resolvedCampaignPath, err := resolveCampaignConfigPath(pipelineCfg, campaignFlag, campaignFileFlag)
if err != nil {
return nil, err
}
campaignCfg, err := config.LoadCampaign(resolvedCampaignPath)
if err != nil {
return nil, err
}
if selectedID := strings.TrimSpace(campaignFlag); selectedID != "" && strings.TrimSpace(campaignFileFlag) == "" {
if got := config.CampaignID(campaignCfg); got != selectedID {
return nil, fmt.Errorf("campaign config %q invalid: campaign_id %q does not match selected campaign %q", resolvedCampaignPath, got, selectedID)
}
}
return &pipelineCampaignConfig{
PipelinePath: resolvedPipelinePath,
CampaignPath: resolvedCampaignPath,
Pipeline: pipelineCfg,
Campaign: campaignCfg,
}, nil
}
func findRemoteSessionConfig(ctx context.Context, store storage.ObjectStore, sessionPrefix, remoteKey string) (storage.ObjectInfo, error) {
objects, err := store.List(ctx, sessionPrefix)
if err != nil {
return storage.ObjectInfo{}, fmt.Errorf("remote session %q list failed: %w", remoteKey, err)
}
for _, obj := range objects {
if obj.Key == remoteKey {
return obj, nil
}
}
return storage.ObjectInfo{}, fmt.Errorf("remote session %q not found", remoteKey)
}
func s3BucketName(cfg *config.PipelineConfig) string {
if cfg == nil || cfg.Storage.S3 == nil {
return ""
}
return strings.TrimSpace(cfg.Storage.S3.Bucket)
}

View File

@@ -0,0 +1,416 @@
package app
import (
"context"
"encoding/json"
"errors"
"fmt"
"os"
"path/filepath"
"strings"
"testing"
"time"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/notarius"
"gitea.maximumdirect.net/eric/narratio/internal/artifactmodel"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
"gitea.maximumdirect.net/eric/narratio/internal/stage"
)
type materializingNotariusRunner struct {
cfg *config.NotariusConfig
requests []notarius.RunRequest
failuresRemaining int
}
func (r *materializingNotariusRunner) Run(_ context.Context, req notarius.RunRequest) (notarius.RunResult, error) {
r.requests = append(r.requests, req)
if r.failuresRemaining > 0 {
r.failuresRemaining--
return notarius.RunResult{}, errors.New("notarius execution failed")
}
externalRunID := fmt.Sprintf("notarius-run-%d", len(r.requests))
bundle := filepath.Join(req.OutputRoot, externalRunID)
lanesDir := filepath.Join(bundle, "lanes")
if err := os.MkdirAll(lanesDir, 0o755); err != nil {
return notarius.RunResult{}, err
}
for path, content := range map[string]string{
filepath.Join(bundle, "index.json"): `{"manifest_file":"manifest.json"}`,
filepath.Join(bundle, "manifest.json"): `{}`,
filepath.Join(bundle, "rejected.json"): `{"rejected":[]}`,
filepath.Join(bundle, "warnings.json"): `{"warnings":[]}`,
filepath.Join(lanesDir, "npcs.json"): `{"npcs":[]}`,
} {
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
return notarius.RunResult{}, err
}
}
output := r.cfg.Outputs["npc_registry"]
return notarius.RunResult{
Receipt: notarius.Receipt{
SchemaVersion: notarius.ReceiptSchemaVersion, RunID: externalRunID,
PipelineID: req.PipelineID, OutputDirectory: bundle, IndexFile: "index.json",
NormalizedOutputCount: 1, ValidationStatus: "valid",
},
BundleRoot: bundle,
Index: notarius.Index{
Path: filepath.Join(bundle, "index.json"), RejectedPath: filepath.Join(bundle, "rejected.json"),
WarningsPath: filepath.Join(bundle, "warnings.json"),
Lanes: []notarius.LaneDescriptor{{
LaneID: output.LaneID, File: "lanes/npcs.json", Path: filepath.Join(lanesDir, "npcs.json"),
MediaType: output.MediaType, SchemaID: output.SchemaID,
SchemaVersion: output.SchemaVersion, ModuleKey: output.ModuleKey,
}},
},
}, nil
}
func TestExtractLifecycleDisabledThenEnabled(t *testing.T) {
cfg, env, runner := extractionLifecycleFixture(t, false)
plan, err := BuildSingleStagePlan("extract")
if err != nil {
t.Fatalf("BuildSingleStagePlan(extract) error = %v", err)
}
first, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env})
if err != nil {
t.Fatalf("disabled executeStages() error = %v", err)
}
if len(first.Executed) != 1 || len(first.Skipped) != 1 || first.Skipped[0] != "extract" || len(runner.requests) != 0 {
t.Fatalf("disabled summary = %#v requests=%d", first, len(runner.requests))
}
cfg.Pipeline.Notarius.Enabled = true
second, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env})
if err != nil {
t.Fatalf("enabled executeStages() error = %v", err)
}
if len(second.Executed) != 1 || len(second.Skipped) != 0 || len(runner.requests) != 1 {
t.Fatalf("enabled summary = %#v requests=%d", second, len(runner.requests))
}
loaded, err := (&manifest.LocalStore{}).Load(context.Background(), second.ManifestPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
if loaded.Stages["extract"] == nil || loaded.Stages["extract"].Status != manifest.StatusSucceeded || len(loaded.Stages["extract"].Outputs) != 2 {
t.Fatalf("extract record = %#v, want succeeded manifest-ready outputs", loaded.Stages["extract"])
}
}
func TestExtractLifecycleChangedOutcomeRerunsSucceededDownstream(t *testing.T) {
cfg, env, runner := extractionLifecycleFixture(t, false)
analyzeRuns := 0
plan := extractionLifecyclePlan(t, &analyzeRuns)
if _, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env}); err != nil {
t.Fatalf("disabled executeStages() error = %v", err)
}
if analyzeRuns != 1 || len(runner.requests) != 0 {
t.Fatalf("disabled run analyze=%d Notarius=%d, want 1 and 0", analyzeRuns, len(runner.requests))
}
cfg.Pipeline.Notarius.Enabled = true
summary, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env})
if err != nil {
t.Fatalf("enabled executeStages() error = %v", err)
}
if analyzeRuns != 2 || len(runner.requests) != 1 {
t.Fatalf("enabled run analyze=%d Notarius=%d, want 2 and 1", analyzeRuns, len(runner.requests))
}
if len(summary.Executed) != 2 || len(summary.Skipped) != 0 {
t.Fatalf("enabled summary = %#v, want extract and analyze executed", summary)
}
}
func TestExtractLifecycleFailureInvalidatesAndOrdinaryRetryRerunsDownstream(t *testing.T) {
cfg, env, runner := extractionLifecycleFixture(t, true)
runner.failuresRemaining = 1
markLifecycleStageSucceeded(t, cfg, "analyze")
analyzeRuns := 0
plan := extractionLifecyclePlan(t, &analyzeRuns)
if _, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env}); err == nil || !strings.Contains(err.Error(), "notarius execution failed") {
t.Fatalf("failed executeStages() error = %v", err)
}
failed := loadLifecycleManifest(t, cfg)
if failed.Stages["extract"].Status != manifest.StatusFailed || failed.Stages["analyze"].Status != manifest.StatusStale {
t.Fatalf("failed lifecycle extract=%#v analyze=%#v", failed.Stages["extract"], failed.Stages["analyze"])
}
if failed.Stages["analyze"].Error == nil || failed.Stages["analyze"].Error.Message != staleReasonFailure {
t.Fatalf("analyze stale reason = %#v, want %q", failed.Stages["analyze"].Error, staleReasonFailure)
}
summary, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env})
if err != nil {
t.Fatalf("retry executeStages() error = %v", err)
}
if analyzeRuns != 1 || len(runner.requests) != 2 || len(summary.Executed) != 2 {
t.Fatalf("retry analyze=%d Notarius=%d summary=%#v", analyzeRuns, len(runner.requests), summary)
}
}
func TestExtractLifecycleForcedSelfSkipInvalidatesDownstream(t *testing.T) {
cfg, env, _ := extractionLifecycleFixture(t, false)
markLifecycleStageSucceeded(t, cfg, "analyze")
plan, _ := BuildSingleStagePlan("extract")
if _, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env, Force: true}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
loaded := loadLifecycleManifest(t, cfg)
if loaded.Stages["extract"].Status != manifest.StatusSkipped || loaded.Stages["analyze"].Status != manifest.StatusStale {
t.Fatalf("forced self-skip extract=%#v analyze=%#v", loaded.Stages["extract"], loaded.Stages["analyze"])
}
if loaded.Stages["analyze"].Error == nil || loaded.Stages["analyze"].Error.Message != staleReasonForcedReplacement {
t.Fatalf("analyze stale reason = %#v, want %q", loaded.Stages["analyze"].Error, staleReasonForcedReplacement)
}
}
func TestExtractLifecycleForcedFailureInvalidatesDownstream(t *testing.T) {
cfg, env, runner := extractionLifecycleFixture(t, true)
plan, _ := BuildSingleStagePlan("extract")
first, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env})
if err != nil {
t.Fatalf("initial executeStages() error = %v", err)
}
succeeded := loadLifecycleManifest(t, cfg).Stages["extract"]
if succeeded == nil || succeeded.Status != manifest.StatusSucceeded || len(succeeded.Outputs) == 0 || len(succeeded.Logs) == 0 || len(succeeded.Metadata) == 0 {
t.Fatalf("initial extraction result = %#v, want succeeded result details", succeeded)
}
markLifecycleStageSucceeded(t, cfg, "analyze")
runner.failuresRemaining = 1
if _, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env, Force: true}); err == nil {
t.Fatal("executeStages() error = nil, want forced extraction failure")
}
loaded := loadLifecycleManifest(t, cfg)
if loaded.Stages["extract"].Status != manifest.StatusFailed || loaded.Stages["analyze"].Status != manifest.StatusStale {
t.Fatalf("forced failure extract=%#v analyze=%#v", loaded.Stages["extract"], loaded.Stages["analyze"])
}
if loaded.Stages["analyze"].Error == nil || loaded.Stages["analyze"].Error.Message != staleReasonForcedReplacement {
t.Fatalf("analyze stale reason = %#v, want %q", loaded.Stages["analyze"].Error, staleReasonForcedReplacement)
}
failed := loaded.Stages["extract"]
if len(failed.Outputs) != 0 || len(failed.Logs) != 0 || len(failed.GeneratedConfigs) != 0 || len(failed.Metadata) != 0 {
t.Fatalf("failed replacement inherited extraction result details: %#v", failed)
}
historical, err := (&manifest.LocalStore{}).LoadRun(context.Background(), first.RunManifestPath)
if err != nil {
t.Fatalf("LoadRun(initial) error = %v", err)
}
historicalExtract := historical.Stages["extract"]
if historicalExtract == nil || historicalExtract.Status != manifest.StatusSucceeded || len(historicalExtract.Outputs) == 0 || len(historicalExtract.Logs) == 0 || len(historicalExtract.Metadata) == 0 {
t.Fatalf("historical extraction result = %#v, want preserved succeeded details", historicalExtract)
}
if _, err := os.Stat(succeeded.Outputs[0].LocalPath); err != nil {
t.Fatalf("durable extraction output was not preserved: %v", err)
}
}
func TestExtractLifecycleRepeatedSelfSkipPreservesSucceededDownstream(t *testing.T) {
cfg, env, runner := extractionLifecycleFixture(t, false)
analyzeRuns := 0
plan := extractionLifecyclePlan(t, &analyzeRuns)
if _, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env}); err != nil {
t.Fatalf("first executeStages() error = %v", err)
}
second, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env})
if err != nil {
t.Fatalf("second executeStages() error = %v", err)
}
if analyzeRuns != 1 || len(runner.requests) != 0 {
t.Fatalf("repeated disabled run analyze=%d Notarius=%d, want 1 and 0", analyzeRuns, len(runner.requests))
}
if len(second.Executed) != 1 || len(second.Skipped) != 2 {
t.Fatalf("second summary = %#v, want executed self-skip and skipped analyze", second)
}
loaded := loadLifecycleManifest(t, cfg)
if loaded.Stages["analyze"].Status != manifest.StatusSucceeded {
t.Fatalf("analyze = %#v, want succeeded", loaded.Stages["analyze"])
}
}
func TestExtractLifecycleSkipsCurrentResumableResult(t *testing.T) {
cfg, env, runner := extractionLifecycleFixture(t, true)
analyzeRuns := 0
plan := extractionLifecyclePlan(t, &analyzeRuns)
if _, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env}); err != nil {
t.Fatalf("first executeStages() error = %v", err)
}
resumed, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env})
if err != nil {
t.Fatalf("resume executeStages() error = %v", err)
}
if len(resumed.Executed) != 0 || len(resumed.Skipped) != 2 || len(runner.requests) != 1 || analyzeRuns != 1 {
t.Fatalf("resume summary = %#v requests=%d analyze=%d", resumed, len(runner.requests), analyzeRuns)
}
}
func TestExtractLifecycleResumesAndRerunsObsoleteResults(t *testing.T) {
for _, test := range []struct {
name string
mutate func(*testing.T, *config.Config, *manifest.Manifest)
}{
{name: "configuration changed", mutate: func(_ *testing.T, cfg *config.Config, _ *manifest.Manifest) {
output := cfg.Pipeline.Notarius.Outputs["npc_registry"]
output.SchemaVersion = "v2"
cfg.Pipeline.Notarius.Outputs["npc_registry"] = output
}},
{name: "payload missing", mutate: func(t *testing.T, _ *config.Config, m *manifest.Manifest) {
if err := os.Remove(m.Stages["extract"].Outputs[0].LocalPath); err != nil {
t.Fatalf("Remove() error = %v", err)
}
}},
{name: "payload tampered", mutate: func(t *testing.T, _ *config.Config, m *manifest.Manifest) {
if err := os.WriteFile(m.Stages["extract"].Outputs[0].LocalPath, []byte(`{"npcs":["tampered"]}`), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
}},
{name: "record incompatible", mutate: func(_ *testing.T, _ *config.Config, m *manifest.Manifest) {
m.Stages["extract"].Outputs[0].Contract.SchemaID = "incompatible"
}},
} {
t.Run(test.name, func(t *testing.T) {
cfg, env, runner := extractionLifecycleFixture(t, true)
plan, _ := BuildSingleStagePlan("extract")
first, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env})
if err != nil {
t.Fatalf("first executeStages() error = %v", err)
}
persisted, err := (&manifest.LocalStore{}).Load(context.Background(), first.ManifestPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
test.mutate(t, cfg, persisted)
if err := (&manifest.LocalStore{}).Save(context.Background(), first.ManifestPath, persisted); err != nil {
t.Fatalf("Save(mutated) error = %v", err)
}
rerun, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env})
if err != nil {
t.Fatalf("rerun executeStages() error = %v", err)
}
if len(rerun.Executed) != 1 || len(rerun.Skipped) != 0 || len(runner.requests) != 2 {
t.Fatalf("rerun summary = %#v requests=%d", rerun, len(runner.requests))
}
})
}
}
func TestExtractLifecycleUnsafeResumeErrorPreservesSuccess(t *testing.T) {
cfg, env, runner := extractionLifecycleFixture(t, true)
analyzeRuns := 0
plan := extractionLifecyclePlan(t, &analyzeRuns)
first, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env})
if err != nil {
t.Fatalf("first executeStages() error = %v", err)
}
store := &manifest.LocalStore{}
persisted, err := store.Load(context.Background(), first.ManifestPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
persisted.Stages["extract"].Outputs[0].LocalPath = filepath.Join(cfg.Pipeline.Workspace.Root, "outside.json")
if err := store.Save(context.Background(), first.ManifestPath, persisted); err != nil {
t.Fatalf("Save() error = %v", err)
}
before, _ := json.Marshal(map[string]*manifest.StageRecord{
"extract": persisted.Stages["extract"],
"analyze": persisted.Stages["analyze"],
})
if _, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env}); err == nil || !strings.Contains(err.Error(), "unsafe") {
t.Fatalf("executeStages() error = %v, want unsafe resume failure", err)
}
afterManifest, err := store.Load(context.Background(), first.ManifestPath)
if err != nil {
t.Fatalf("Load(after) error = %v", err)
}
after, _ := json.Marshal(map[string]*manifest.StageRecord{
"extract": afterManifest.Stages["extract"],
"analyze": afterManifest.Stages["analyze"],
})
if string(before) != string(after) || len(runner.requests) != 1 || analyzeRuns != 1 {
t.Fatalf("successful records changed: before=%s after=%s requests=%d analyze=%d", before, after, len(runner.requests), analyzeRuns)
}
}
func extractionLifecyclePlan(t *testing.T, analyzeRuns *int) []stage.Stage {
t.Helper()
plan, err := BuildSingleStagePlan("extract")
if err != nil {
t.Fatalf("BuildSingleStagePlan(extract) error = %v", err)
}
return append(plan, countingStage{name: "analyze", runs: analyzeRuns})
}
func loadLifecycleManifest(t *testing.T, cfg *config.Config) *manifest.Manifest {
t.Helper()
loaded, err := (&manifest.LocalStore{}).Load(context.Background(), manifestPathFor(cfg))
if err != nil {
t.Fatalf("Load() error = %v", err)
}
return loaded
}
func markLifecycleStageSucceeded(t *testing.T, cfg *config.Config, name string) {
t.Helper()
loaded := loadLifecycleManifest(t, cfg)
loaded.MarkStageSucceeded(name, time.Now().UTC(), nil)
if err := (&manifest.LocalStore{}).Save(context.Background(), manifestPathFor(cfg), loaded); err != nil {
t.Fatalf("Save() error = %v", err)
}
}
func extractionLifecycleFixture(t *testing.T, enabled bool) (*config.Config, *stage.Env, *materializingNotariusRunner) {
t.Helper()
cfg := testConfig(t)
root := cfg.Pipeline.Workspace.Root
binary := filepath.Join(root, "notarius")
configPath := filepath.Join(root, "notarius.yml")
workingDirectory := filepath.Join(root, "notarius-work")
if err := os.WriteFile(binary, []byte("#!/bin/sh\nexit 0\n"), 0o755); err != nil {
t.Fatalf("WriteFile(binary) error = %v", err)
}
if err := os.WriteFile(configPath, []byte("pipelines: {}\n"), 0o644); err != nil {
t.Fatalf("WriteFile(config) error = %v", err)
}
if err := os.Mkdir(workingDirectory, 0o755); err != nil {
t.Fatalf("Mkdir(working directory) error = %v", err)
}
cfg.Pipeline.Notarius = &config.NotariusConfig{
Enabled: enabled, Binary: binary, ConfigPath: configPath, PipelineID: "dnd-session",
Timeout: "45m", WorkingDirectory: workingDirectory,
Outputs: map[string]config.NotariusOutputConfig{
"npc_registry": {
LaneID: "npc-registry", MediaType: "application/json", SchemaID: "notarius.dnd.npc_registry",
SchemaVersion: "v1", ModuleKey: "dnd/npc-registry",
},
},
}
paths, err := artifacts.NewLocalStore(root).EnsureLayoutFor(cfg.Session.Campaign, cfg.Session.SessionID)
if err != nil {
t.Fatalf("EnsureLayoutFor() error = %v", err)
}
inputPath := filepath.Join(paths.ArtifactsDir, "trimmed.from-manifest.json")
if err := os.WriteFile(inputPath, []byte(`{"segments":[]}`), 0o644); err != nil {
t.Fatalf("WriteFile(input) error = %v", err)
}
m := manifest.New(cfg.Session.SessionID, time.Now().UTC())
m.Campaign = cfg.Session.Campaign
m.MarkStageSucceeded("trim", time.Now().UTC(), []manifest.ArtifactRecord{{
Kind: artifactmodel.TranscriptOutputKindFinalTrimmed, SourceID: artifactmodel.SourceTranscriptFinalTrimmed,
LocalPath: inputPath,
}})
if err := (&manifest.LocalStore{}).Save(context.Background(), paths.ManifestPath, m); err != nil {
t.Fatalf("Save(seed) error = %v", err)
}
runner := &materializingNotariusRunner{cfg: cfg.Pipeline.Notarius}
return cfg, &stage.Env{Notarius: runner}, runner
}

View File

@@ -0,0 +1,21 @@
package app
import (
"context"
"fmt"
"log/slog"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func newCommandObjectStore(ctx context.Context, cfg *config.Config, logger *slog.Logger) (storage.ObjectStore, error) {
if _, err := loadSecretsFromConfig(cfg, logger); err != nil {
return nil, fmt.Errorf("load secrets from files: %w", err)
}
store, err := newObjectStoreFromConfigFn(ctx, cfg)
if err != nil {
return nil, fmt.Errorf("initialize object store backend: %w", err)
}
return store, nil
}

View File

@@ -0,0 +1,165 @@
package app
import (
"context"
"errors"
"os"
"path/filepath"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestNewCommandObjectStoreLoadsSecretsBeforeFactory(t *testing.T) {
accessKeyEnv := "NARRATIO_TEST_COMMAND_STORE_KEY_ID"
secretKeyEnv := "NARRATIO_TEST_COMMAND_STORE_SECRET"
restoreEnvAfterTest(t, accessKeyEnv, secretKeyEnv)
secretsDir := t.TempDir()
mustWriteSecretFile(t, filepath.Join(secretsDir, accessKeyEnv), "loaded-key-id\n")
mustWriteSecretFile(t, filepath.Join(secretsDir, secretKeyEnv), "loaded-secret\n")
cfg := commandObjectStoreTestConfig(secretsDir)
fake := &storage.FakeBackend{}
called := false
origStoreFn := newObjectStoreFromConfigFn
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
called = true
if got := os.Getenv(accessKeyEnv); got != "loaded-key-id" {
return nil, errors.New("access key was not loaded before object store init")
}
if got := os.Getenv(secretKeyEnv); got != "loaded-secret" {
return nil, errors.New("secret key was not loaded before object store init")
}
return fake, nil
}
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
})
store, err := newCommandObjectStore(context.Background(), cfg, nil)
if err != nil {
t.Fatalf("newCommandObjectStore() error = %v", err)
}
if store != fake {
t.Fatalf("store = %#v, want fake backend", store)
}
if !called {
t.Fatal("object store factory was not called")
}
}
func TestNewCommandObjectStorePreservesExistingEnv(t *testing.T) {
accessKeyEnv := "NARRATIO_TEST_COMMAND_STORE_EXISTING_KEY_ID"
secretKeyEnv := "NARRATIO_TEST_COMMAND_STORE_EXISTING_SECRET"
t.Setenv(accessKeyEnv, "existing-key-id")
t.Setenv(secretKeyEnv, "existing-secret")
secretsDir := t.TempDir()
mustWriteSecretFile(t, filepath.Join(secretsDir, accessKeyEnv), "file-key-id\n")
mustWriteSecretFile(t, filepath.Join(secretsDir, secretKeyEnv), "file-secret\n")
cfg := commandObjectStoreTestConfig(secretsDir)
origStoreFn := newObjectStoreFromConfigFn
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
if got := os.Getenv(accessKeyEnv); got != "existing-key-id" {
return nil, errors.New("existing access key was overwritten")
}
if got := os.Getenv(secretKeyEnv); got != "existing-secret" {
return nil, errors.New("existing secret key was overwritten")
}
return &storage.FakeBackend{}, nil
}
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
})
if _, err := newCommandObjectStore(context.Background(), cfg, nil); err != nil {
t.Fatalf("newCommandObjectStore() error = %v", err)
}
}
func TestNewCommandObjectStoreSecretErrorStopsFactory(t *testing.T) {
cfg := commandObjectStoreTestConfig(filepath.Join(t.TempDir(), "missing"))
called := false
origStoreFn := newObjectStoreFromConfigFn
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
called = true
return &storage.FakeBackend{}, nil
}
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
})
_, err := newCommandObjectStore(context.Background(), cfg, nil)
if err == nil {
t.Fatal("expected error, got nil")
}
if called {
t.Fatal("object store factory was called after secret load failure")
}
if !strings.Contains(err.Error(), "load secrets from files") {
t.Fatalf("error = %q, want secret loading context", err.Error())
}
}
func TestNewCommandObjectStoreFactoryErrorIsContextual(t *testing.T) {
cfg := commandObjectStoreTestConfig("")
origStoreFn := newObjectStoreFromConfigFn
newObjectStoreFromConfigFn = func(context.Context, *config.Config) (storage.ObjectStore, error) {
return nil, errors.New("factory boom")
}
t.Cleanup(func() {
newObjectStoreFromConfigFn = origStoreFn
})
_, err := newCommandObjectStore(context.Background(), cfg, nil)
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "initialize object store backend") || !strings.Contains(err.Error(), "factory boom") {
t.Fatalf("error = %q, want factory context", err.Error())
}
}
func commandObjectStoreTestConfig(secretsDir string) *config.Config {
cfg := &config.Config{
Pipeline: &config.PipelineConfig{
Storage: config.StorageConfig{
Backend: "s3",
S3: &config.StorageS3Config{
Bucket: "test-bucket",
AccessKeyIDEnv: "NARRATIO_TEST_COMMAND_STORE_KEY_ID",
SecretKeyEnv: "NARRATIO_TEST_COMMAND_STORE_SECRET",
},
},
},
}
if strings.TrimSpace(secretsDir) != "" {
cfg.Pipeline.Secrets = &config.SecretsConfig{EnvDir: secretsDir}
}
return cfg
}
func restoreEnvAfterTest(t *testing.T, names ...string) {
t.Helper()
originals := make(map[string]string, len(names))
present := make(map[string]bool, len(names))
for _, name := range names {
value, ok := os.LookupEnv(name)
originals[name] = value
present[name] = ok
_ = os.Unsetenv(name)
}
t.Cleanup(func() {
for _, name := range names {
if present[name] {
_ = os.Setenv(name, originals[name])
} else {
_ = os.Unsetenv(name)
}
}
})
}

View File

@@ -0,0 +1,177 @@
package app
import (
"context"
"fmt"
"io"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifactpolicy"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
func buildHelperArtifactCatalog(cfg *config.Config, m *manifest.Manifest) (*artifacts.ArtifactCatalog, error) {
catalog := artifacts.NewArtifactCatalog()
if err := catalog.RegisterBuiltIns(); err != nil {
return nil, err
}
configured := map[string]artifacts.ConfiguredArtifactDefinition{}
if cfg.Pipeline.Scriptorium != nil {
for key, item := range cfg.Pipeline.Scriptorium.Artifacts {
configured[key] = artifacts.ConfiguredArtifactDefinition{Enabled: item.Enabled, OutputPath: item.OutputPath}
}
}
if err := catalog.RegisterConfiguredArtifacts(configured, nil); err != nil {
return nil, err
}
extractionDefinitions := artifacts.ExtractionDefinitionsFromConfig(cfg.Pipeline.Notarius)
if err := catalog.RegisterExtractionArtifacts(extractionDefinitions); err != nil {
return nil, err
}
paths := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root).SessionPathsFor(cfg.Session.Campaign, cfg.Session.SessionID)
if cfg.Pipeline.Notarius != nil && cfg.Pipeline.Notarius.Enabled {
catalog.HydrateExtractionArtifacts(paths, m, extractionDefinitions)
}
return catalog, nil
}
func writeArtifactList(out io.Writer, cfg *config.Config, catalog *artifacts.ArtifactCatalog, locks *effectiveLocks, publishedRemoteState map[string]string) {
lockSet := lockSourceSet(locks.All)
fmt.Fprintln(out, "Built-in:")
for _, transcript := range artifacts.RuntimeTranscriptArtifacts() {
writeArtifactLine(out, transcript.SourceID, lockSet)
}
writeArtifactLine(out, artifacts.ArtifactBoundsSession, lockSet)
fmt.Fprintln(out, "Configured:")
for _, entry := range catalog.ListConfigured() {
writeArtifactLine(out, entry.SourceID, lockSet)
}
fmt.Fprintln(out, "Extraction:")
for _, entry := range catalog.ListExtraction() {
state := "unavailable"
if entry.Available {
state = "available"
}
writeExtractionArtifactLine(out, entry.SourceID, state, entry.Provenance, lockSet)
}
fmt.Fprintln(out, "Previous-session:")
for _, req := range artifacts.CollectPreviousArtifactRequirements(configuredScriptoriumArtifacts(cfg)) {
fmt.Fprintf(out, "- %s required=%t\n", artifactpolicy.PreviousSessionSourceID(req.Name), req.Required)
}
fmt.Fprintln(out, "Published:")
for _, rule := range cfg.Pipeline.Publish.Outputs {
writePublishedOutputLine(out, rule, catalog, lockSet, publishedRemoteState)
}
}
func writeExtractionArtifactLine(out io.Writer, source, state, provenance string, lockSet map[string]config.PublishLockRule) {
parts := []string{source, "planned", state}
if strings.TrimSpace(provenance) != "" {
parts = append(parts, "provenance="+strings.TrimSpace(provenance))
}
if _, ok := lockSet[source]; ok {
parts = append(parts, "locked")
}
fmt.Fprintf(out, "- %s\n", strings.Join(parts, " "))
}
func writeArtifactLine(out io.Writer, source string, lockSet map[string]config.PublishLockRule) {
parts := []string{source}
if _, ok := lockSet[source]; ok {
parts = append(parts, "locked")
}
fmt.Fprintf(out, "- %s\n", strings.Join(parts, " "))
}
func writePublishedOutputLine(out io.Writer, rule config.PublishOutputRule, catalog *artifacts.ArtifactCatalog, lockSet map[string]config.PublishLockRule, remoteState map[string]string) {
source := strings.TrimSpace(rule.Source)
parts := []string{source}
if _, ok := lockSet[source]; ok {
parts = append(parts, "locked")
}
dest, showDest, err := helperPublishedOutputDest(rule, catalog)
if err != nil {
parts = append(parts, "remote=error")
fmt.Fprintf(out, "- %s\n", strings.Join(parts, " "))
return
}
if showDest {
parts = append(parts, "dest="+dest)
}
if state := remoteState[publishedOutputRemoteStateKey(source, dest)]; state != "" {
parts = append(parts, state)
}
fmt.Fprintf(out, "- %s\n", strings.Join(parts, " "))
}
func remotePublishedOutputAvailability(ctx context.Context, cfg *config.Config, store storage.ObjectStore, catalog *artifacts.ArtifactCatalog) map[string]string {
out := map[string]string{}
sessionPrefix := artifacts.S3SessionPrefix(cfg.Pipeline.Storage.S3.RootPrefix, cfg.Session.Campaign, cfg.Session.SessionID)
for _, rule := range cfg.Pipeline.Publish.Outputs {
source := strings.TrimSpace(rule.Source)
dest, _, err := helperPublishedOutputDest(rule, catalog)
if err != nil {
out[publishedOutputRemoteStateKey(source, "")] = "remote=error"
continue
}
key := artifacts.S3PublishedOutputKey(sessionPrefix, dest)
if exists, err := store.Exists(ctx, key); err == nil && exists {
out[publishedOutputRemoteStateKey(source, dest)] = "remote=published"
} else if err != nil {
out[publishedOutputRemoteStateKey(source, dest)] = "remote=error"
} else {
out[publishedOutputRemoteStateKey(source, dest)] = "remote=missing"
}
}
return out
}
func helperPublishedOutputDest(rule config.PublishOutputRule, catalog *artifacts.ArtifactCatalog) (string, bool, error) {
source := strings.TrimSpace(rule.Source)
normalized, err := artifactpolicy.ResolvePublishedDestinationWithExtractions(
source,
rule.Dest,
helperConfiguredOutputPathMap(catalog),
helperExtractionOutputSet(catalog),
)
if err != nil {
return "", false, err
}
entry, ok := catalog.Lookup(source)
showDest := !ok || strings.TrimSpace(entry.CanonicalRelPath) != normalized
return normalized, showDest, nil
}
func helperExtractionOutputSet(catalog *artifacts.ArtifactCatalog) map[string]struct{} {
out := map[string]struct{}{}
if catalog == nil {
return out
}
for _, entry := range catalog.ListExtraction() {
if strings.TrimSpace(entry.ExtractionKey) != "" {
out[entry.ExtractionKey] = struct{}{}
}
}
return out
}
func helperConfiguredOutputPathMap(catalog *artifacts.ArtifactCatalog) map[string]string {
out := map[string]string{}
if catalog == nil {
return out
}
for _, entry := range catalog.ListConfigured() {
if strings.TrimSpace(entry.ConfiguredKey) == "" {
continue
}
out[entry.ConfiguredKey] = strings.TrimSpace(entry.CanonicalRelPath)
}
return out
}
func publishedOutputRemoteStateKey(source, dest string) string {
return strings.TrimSpace(source) + "\x00" + strings.TrimSpace(dest)
}

View File

@@ -0,0 +1,39 @@
package app
import (
"context"
"flag"
"fmt"
"io"
"strings"
)
// ArtifactsList lists effective artifact sources.
func ArtifactsList(ctx context.Context, args []string, out io.Writer) error {
fs := flag.NewFlagSet("artifacts list", flag.ContinueOnError)
fs.SetOutput(io.Discard)
var flags commonConfigFlags
var remote bool
addCommonConfigFlags(fs, &flags)
fs.BoolVar(&remote, "remote", false, "inspect remote publish availability")
if err := parseSessionAwareFlags("artifacts list", fs, args, &flags.sessionID); err != nil {
return err
}
if strings.TrimSpace(flags.sessionID) == "" {
return fmt.Errorf("artifacts list: session_id is required")
}
cfg, store, locks, m, err := loadHelperContext(ctx, flags, remote)
if err != nil {
return fmt.Errorf("artifacts list: %w", err)
}
catalog, err := buildHelperArtifactCatalog(cfg, m)
if err != nil {
return fmt.Errorf("artifacts list: %w", err)
}
publishedRemoteState := map[string]string{}
if remote && store != nil {
publishedRemoteState = remotePublishedOutputAvailability(ctx, cfg, store, catalog)
}
writeArtifactList(out, cfg, catalog, locks, publishedRemoteState)
return nil
}

View File

@@ -0,0 +1,140 @@
package app
import (
"context"
"fmt"
"io"
"os"
"path/filepath"
"sort"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
type finding struct {
Severity string
Category string
Message string
}
type findingError struct {
count int
}
func (e findingError) Error() string {
return fmt.Sprintf("%d validation error(s)", e.count)
}
func renderFindings(out io.Writer, campaign, sessionID string, findings []finding) error {
if campaign != "" || sessionID != "" {
fmt.Fprintf(out, "Campaign: %s\n", campaign)
fmt.Fprintf(out, "Session: %s\n\n", sessionID)
}
errorsCount := 0
for _, f := range findings {
if f.Severity == "ERROR" {
errorsCount++
}
fmt.Fprintf(out, "%-5s %-10s %s\n", f.Severity, f.Category, f.Message)
}
if errorsCount > 0 {
return findingError{count: errorsCount}
}
return nil
}
func okFinding(category, msg string) finding { return finding{"OK", category, msg} }
func infoFinding(category, msg string) finding { return finding{"INFO", category, msg} }
func warnFinding(category, msg string) finding { return finding{"WARN", category, msg} }
func errorFinding(category, msg string) finding { return finding{"ERROR", category, msg} }
func sessionSourceSummary(cfg *config.Config) string {
source := cfg.SessionSource.Source
if source == "" {
source = "session_config"
}
if cfg.SessionSource.S3Key != "" {
return source + " " + cfg.SessionSource.S3Key
}
return source + " " + cfg.SessionPath
}
func validateStableInputFindings(cfg *config.Config) []finding {
checks := inspectStableInputs(cfg)
out := make([]finding, 0, len(checks))
for _, check := range checks {
if check.Err != nil {
msg := check.Name + ": " + check.Err.Error()
if strings.TrimSpace(check.Path) != "" {
msg = fmt.Sprintf("%s missing: %v", check.Name, check.Err)
}
out = append(out, errorFinding("inputs", msg))
continue
}
out = append(out, okFinding("inputs", check.Name+": "+check.Path))
}
return out
}
func resolveHelperConfigRelativePath(input config.ResolvedInputFile) (string, error) {
if strings.TrimSpace(input.ConfigPath) == "" {
return "", fmt.Errorf("source config path is required")
}
path := strings.TrimSpace(input.Path)
if path == "" {
return "", fmt.Errorf("path is required")
}
if filepath.IsAbs(path) {
return filepath.Clean(path), nil
}
return filepath.Clean(filepath.Join(filepath.Dir(input.ConfigPath), path)), nil
}
func validateLocalAudioFindings(cfg *config.Config) []finding {
check := inspectLocalAudioPresence(cfg)
if !check.Checked {
return nil
}
if check.Err != nil {
return []finding{errorFinding("audio", check.Err.Error())}
}
return []finding{okFinding("audio", fmt.Sprintf("%d local audio file(s)", len(check.Paths)))}
}
func validateRemoteAudioFinding(ctx context.Context, cfg *config.Config, store storage.ObjectStore) finding {
check := inspectRemoteAudioPresence(ctx, cfg, store)
if check.Err != nil {
return errorFinding("audio", check.Err.Error())
}
return okFinding("audio", fmt.Sprintf("%d remote .flac object(s)", len(check.Keys)))
}
func loadLocalManifest(ctx context.Context, path string) (*manifest.Manifest, error) {
if _, err := os.Stat(path); err != nil {
if os.IsNotExist(err) {
return nil, nil
}
return nil, err
}
store := &manifest.LocalStore{}
return store.Load(ctx, path)
}
func writeStageStatuses(out io.Writer, m *manifest.Manifest) {
if m == nil || len(m.Stages) == 0 {
fmt.Fprintln(out, "stages: no stages recorded")
return
}
fmt.Fprintln(out, "stages:")
names := make([]string, 0, len(m.Stages))
for name := range m.Stages {
names = append(names, name)
}
sort.Strings(names)
for _, name := range names {
fmt.Fprintf(out, "- %s: %s\n", name, m.Stages[name].Status)
}
}

View File

@@ -0,0 +1,131 @@
package app
import (
"context"
"flag"
"fmt"
"io"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
type commonConfigFlags struct {
pipelinePath string
campaignPath string
campaignFilePath string
sessionPath string
sessionID string
previousSessionID string
}
func addCommonConfigFlags(fs *flag.FlagSet, flags *commonConfigFlags) {
fs.StringVar(&flags.pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
fs.StringVar(&flags.campaignPath, "campaign", "", "campaign ID")
fs.StringVar(&flags.campaignFilePath, "campaign-file", "", "path to campaign.yml")
fs.StringVar(&flags.sessionPath, "session", "", "path to session.yml")
fs.StringVar(&flags.sessionID, "session-id", "", "session identifier")
fs.StringVar(&flags.previousSessionID, "previous-session-id", "", "expected previous session identifier")
}
func (f commonConfigFlags) sessionOptions() config.SessionLoadOptions {
return config.SessionLoadOptions{
SessionID: f.sessionID,
PreviousSessionID: f.previousSessionID,
}
}
// Session dispatches session helper subcommands.
func Session(ctx context.Context, args []string, out io.Writer) error {
if len(args) == 0 {
return fmt.Errorf("session: expected subcommand: init|validate|status|plan|restore|artifacts|locks")
}
switch args[0] {
case "init":
return SessionInit(ctx, args[1:], out)
case "validate":
return SessionValidate(ctx, args[1:], out)
case "status":
return Status(ctx, args[1:], out)
case "plan":
return Plan(ctx, args[1:], out)
case "restore":
return Restore(ctx, args[1:], out)
case "artifacts":
return ArtifactsList(ctx, args[1:], out)
case "locks":
return SessionLocks(ctx, args[1:], out)
default:
return fmt.Errorf("session: unknown subcommand %q", args[0])
}
}
// SessionLocks dispatches session-oriented publish lock list and mutation
// helpers while preserving the existing lock implementations.
func SessionLocks(ctx context.Context, args []string, out io.Writer) error {
if len(args) > 0 && !isCLIFlagToken(args[0]) {
switch args[0] {
case "add":
return LocksAdd(ctx, args[1:], out)
case "remove":
return LocksRemove(ctx, args[1:], out)
}
}
return LocksList(ctx, args, out)
}
// Artifacts dispatches artifact helper subcommands.
func Artifacts(ctx context.Context, args []string, out io.Writer) error {
if len(args) == 0 {
return fmt.Errorf("artifacts: expected subcommand: list")
}
switch args[0] {
case "list":
return ArtifactsList(ctx, args[1:], out)
default:
return fmt.Errorf("artifacts: unknown subcommand %q", args[0])
}
}
func loadHelperContext(ctx context.Context, flags commonConfigFlags, needStore bool) (*config.Config, storage.ObjectStore, *effectiveLocks, *manifest.Manifest, error) {
cfg, err := loadCommandConfig(ctx, flags.pipelinePath, flags.campaignPath, flags.campaignFilePath, flags.sessionPath, flags.sessionOptions())
if err != nil {
return nil, nil, nil, nil, err
}
if err := config.Validate(cfg); err != nil {
return nil, nil, nil, nil, err
}
var store storage.ObjectStore
if needStore {
store, err = newCommandObjectStore(ctx, cfg, nil)
if err != nil {
return nil, nil, nil, nil, err
}
} else {
store, _ = objectStoreIfConfigured(ctx, cfg)
}
locks, err := loadEffectiveLocks(ctx, cfg, store)
if err != nil {
return nil, nil, nil, nil, err
}
paths := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root).SessionPathsFor(cfg.Session.Campaign, cfg.Session.SessionID)
m, err := loadLocalManifest(ctx, paths.ManifestPath)
if err != nil {
return nil, nil, nil, nil, err
}
return cfg, store, locks, m, nil
}
func objectStoreIfConfigured(ctx context.Context, cfg *config.Config) (storage.ObjectStore, error) {
if cfg == nil || cfg.Pipeline == nil || cfg.Pipeline.Storage.S3 == nil || strings.TrimSpace(cfg.Pipeline.Storage.S3.Bucket) == "" {
return nil, nil
}
store, err := newCommandObjectStore(ctx, cfg, nil)
if err != nil {
return nil, err
}
return store, nil
}

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,276 @@
package app
import (
"context"
"fmt"
"os"
"path"
"path/filepath"
"sort"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
type stableInputCheck struct {
Name string
Path string
Err error
}
type localAudioCheck struct {
Checked bool
Paths []string
Err error
}
type remoteAudioCheck struct {
Checked bool
Prefix string
Keys []string
Err error
}
type previousArtifactReadiness struct {
Requirements []artifacts.PreviousArtifactRequirement
MissingID bool
Err error
}
type remoteCurrentStateCheck struct {
State *RemoteCurrentState
Err error
}
type effectiveLocksCheck struct {
Locks *effectiveLocks
Err error
}
func inspectStableInputs(cfg *config.Config) []stableInputCheck {
items := []struct {
name string
in config.ResolvedInputFile
}{
{name: "speakers", in: cfg.StableInputs.SpeakersFile},
{name: "autocorrect", in: cfg.StableInputs.AutocorrectFile},
{name: "glossary", in: cfg.StableInputs.GlossaryFile},
{name: "players", in: cfg.StableInputs.PlayersFile},
{name: "party", in: cfg.StableInputs.PartyFile},
}
out := make([]stableInputCheck, 0, len(items))
for _, item := range items {
path, err := resolveHelperConfigRelativePath(item.in)
if err != nil {
out = append(out, stableInputCheck{Name: item.name, Err: err})
continue
}
if _, err := os.Stat(path); err != nil {
out = append(out, stableInputCheck{Name: item.name, Path: path, Err: err})
continue
}
out = append(out, stableInputCheck{Name: item.name, Path: path})
}
return out
}
func inspectLocalAudioPresence(cfg *config.Config) localAudioCheck {
if cfg.Session.Inputs.AudioS3 != nil {
return localAudioCheck{}
}
sessionDir := filepath.Dir(cfg.SessionPath)
resolved, err := resolveLocalInspectionAudioPaths(sessionDir, cfg.Session.Inputs)
if err != nil {
return localAudioCheck{Checked: true, Err: err}
}
return localAudioCheck{
Checked: true,
Paths: resolved,
}
}
func inspectRemoteAudioPresence(ctx context.Context, cfg *config.Config, store storage.ObjectStore) remoteAudioCheck {
if cfg.Session.Inputs.AudioS3 == nil {
return remoteAudioCheck{}
}
if store == nil {
return remoteAudioCheck{Checked: true, Err: fmt.Errorf("storage backend is required for remote audio checks")}
}
sessionPrefix := artifacts.S3SessionPrefix(cfg.Pipeline.Storage.S3.RootPrefix, cfg.Session.Campaign, cfg.Session.SessionID)
audioPrefix := artifacts.S3AudioPrefix(sessionPrefix, cfg.Session.Inputs.AudioS3.Prefix)
objects, err := store.List(ctx, audioPrefix)
if err != nil {
return remoteAudioCheck{Checked: true, Prefix: audioPrefix, Err: err}
}
keys := make([]string, 0, len(objects))
seenBase := map[string]string{}
for _, obj := range objects {
key := strings.TrimSpace(obj.Key)
if key == "" || strings.HasSuffix(key, "/") || !isInspectionFlacPath(key) {
continue
}
base := path.Base(key)
if prev, exists := seenBase[base]; exists && prev != key {
return remoteAudioCheck{
Checked: true,
Prefix: audioPrefix,
Err: fmt.Errorf("duplicate s3 audio basename %q from %q and %q", base, prev, key),
}
}
seenBase[base] = key
keys = append(keys, key)
}
sort.Strings(keys)
if len(keys) == 0 {
return remoteAudioCheck{
Checked: true,
Prefix: audioPrefix,
Err: fmt.Errorf("no .flac files found under s3 audio prefix %q", audioPrefix),
}
}
return remoteAudioCheck{
Checked: true,
Prefix: audioPrefix,
Keys: keys,
}
}
func inspectPreviousArtifactReadiness(
ctx context.Context,
cfg *config.Config,
store storage.ObjectStore,
requirements []artifacts.PreviousArtifactRequirement,
) previousArtifactReadiness {
out := previousArtifactReadiness{
Requirements: append([]artifacts.PreviousArtifactRequirement(nil), requirements...),
}
if len(requirements) == 0 {
return out
}
if strings.TrimSpace(cfg.Session.PreviousSessionID) == "" {
out.MissingID = true
return out
}
if store == nil {
out.Err = fmt.Errorf("previous-session artifacts cannot be checked because storage is unavailable")
return out
}
prefix := artifacts.S3SessionPrefix(cfg.Pipeline.Storage.S3.RootPrefix, cfg.Session.Campaign, cfg.Session.PreviousSessionID)
if _, err := artifacts.LoadCurrentState(ctx, store, prefix, artifacts.CurrentStateValidation{
ExpectedSessionID: strings.TrimSpace(cfg.Session.PreviousSessionID),
ExpectedCampaign: strings.TrimSpace(cfg.Session.Campaign),
ValidateRunID: true,
}); err != nil {
out.Err = fmt.Errorf("remote %v", err)
}
return out
}
func inspectRemoteCurrentState(ctx context.Context, cfg *config.Config, store storage.ObjectStore) remoteCurrentStateCheck {
if store == nil {
return remoteCurrentStateCheck{}
}
current, err := discoverRemoteCurrentStateFn(ctx, cfg, store)
if err != nil {
return remoteCurrentStateCheck{Err: err}
}
return remoteCurrentStateCheck{State: current}
}
func inspectEffectiveLocks(ctx context.Context, cfg *config.Config, store storage.ObjectStore) effectiveLocksCheck {
locks, err := loadEffectiveLocks(ctx, cfg, store)
if err != nil {
return effectiveLocksCheck{Err: err}
}
return effectiveLocksCheck{Locks: locks}
}
func resolveLocalInspectionAudioPaths(sessionDir string, inputs config.SessionInputsConfig) ([]string, error) {
if len(inputs.AudioFiles) > 0 {
out := make([]string, 0, len(inputs.AudioFiles))
seenBase := map[string]string{}
for _, item := range inputs.AudioFiles {
resolved, err := resolveInspectionPath(sessionDir, item)
if err != nil {
return nil, err
}
if !isInspectionFlacPath(resolved) {
return nil, fmt.Errorf("audio file %q must have .flac extension", resolved)
}
if err := requireInspectionFile(resolved, "audio file"); err != nil {
return nil, err
}
base := filepath.Base(resolved)
if prev, exists := seenBase[base]; exists && prev != resolved {
return nil, fmt.Errorf("duplicate audio basename %q from %q and %q", base, prev, resolved)
}
seenBase[base] = resolved
out = append(out, resolved)
}
sort.Strings(out)
return out, nil
}
audioDir, err := resolveInspectionPath(sessionDir, inputs.AudioDir)
if err != nil {
return nil, err
}
entries, err := os.ReadDir(audioDir)
if err != nil {
return nil, fmt.Errorf("read audio directory %q: %w", audioDir, err)
}
out := make([]string, 0, len(entries))
for _, entry := range entries {
if entry.IsDir() {
continue
}
full := filepath.Join(audioDir, entry.Name())
if !isInspectionFlacPath(full) {
continue
}
if err := requireInspectionFile(full, "audio file"); err != nil {
return nil, err
}
out = append(out, full)
}
if len(out) == 0 {
return nil, fmt.Errorf("no .flac files found in audio directory %q", audioDir)
}
sort.Strings(out)
return out, nil
}
func resolveInspectionPath(baseDir, inputPath string) (string, error) {
pathValue := strings.TrimSpace(inputPath)
if pathValue == "" {
return "", fmt.Errorf("path is required")
}
if filepath.IsAbs(pathValue) {
return filepath.Clean(pathValue), nil
}
return filepath.Clean(filepath.Join(baseDir, pathValue)), nil
}
func requireInspectionFile(path, label string) error {
info, err := os.Stat(path)
if err != nil {
if os.IsNotExist(err) {
return fmt.Errorf("%s %q does not exist", label, path)
}
return fmt.Errorf("stat %s %q: %w", label, path, err)
}
if info.IsDir() {
return fmt.Errorf("%s %q is a directory", label, path)
}
return nil
}
func isInspectionFlacPath(path string) bool {
return strings.EqualFold(filepath.Ext(strings.TrimSpace(path)), ".flac")
}

View File

@@ -0,0 +1,172 @@
package app
import (
"context"
"flag"
"fmt"
"io"
"sort"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
// Locks dispatches publish lock list and mutation helpers.
func Locks(ctx context.Context, args []string, out io.Writer) error {
if len(args) > 0 && !strings.HasPrefix(args[0], "-") {
switch args[0] {
case "add":
return LocksAdd(ctx, args[1:], out)
case "remove":
return LocksRemove(ctx, args[1:], out)
default:
return fmt.Errorf("locks: unknown subcommand %q", args[0])
}
}
return LocksList(ctx, args, out)
}
// LocksList lists effective publish locks.
func LocksList(ctx context.Context, args []string, out io.Writer) error {
fs := flag.NewFlagSet("locks", flag.ContinueOnError)
fs.SetOutput(io.Discard)
var flags commonConfigFlags
addCommonConfigFlags(fs, &flags)
if err := parseSessionAwareFlags("locks", fs, args, &flags.sessionID); err != nil {
return err
}
if strings.TrimSpace(flags.sessionID) == "" {
return fmt.Errorf("locks: session_id is required")
}
cfg, _, locks, _, err := loadHelperContext(ctx, flags, true)
if err != nil {
return fmt.Errorf("locks: %w", err)
}
writeLocks(out, cfg, locks)
return nil
}
// LocksAdd adds or updates one remote lock.
func LocksAdd(ctx context.Context, args []string, out io.Writer) error {
fs := flag.NewFlagSet("locks add", flag.ContinueOnError)
fs.SetOutput(io.Discard)
var flags commonConfigFlags
var reason string
var force bool
addCommonConfigFlags(fs, &flags)
fs.StringVar(&reason, "reason", "", "lock reason")
fs.BoolVar(&force, "force", false, "update existing remote lock")
source, err := parseSessionIDAndOnePositionalArg("locks add", "source id", fs, args, &flags.sessionID)
if err != nil {
return err
}
if strings.TrimSpace(flags.sessionID) == "" {
return fmt.Errorf("locks add: session_id is required")
}
cfg, store, locks, _, err := loadHelperContext(ctx, flags, true)
if err != nil {
return fmt.Errorf("locks add: %w", err)
}
if _, err := config.ValidatePublishLockRules([]config.PublishLockRule{{Source: source}}, cfg.Pipeline.Scriptorium, cfg.Pipeline.Notarius, "locks add"); err != nil {
return fmt.Errorf("locks add: %w", err)
}
if _, ok := lockSourceSet(locks.Static)[source]; ok {
return fmt.Errorf("locks add: source %q is locked by pipeline config and cannot be modified remotely", source)
}
remoteSet := lockSourceSet(locks.Remote)
if _, exists := remoteSet[source]; exists && !force {
return fmt.Errorf("locks add: remote lock for %q already exists; pass --force to update", source)
}
remoteSet[source] = config.PublishLockRule{Source: source, Reason: strings.TrimSpace(reason)}
remoteLocks := lockMapValues(remoteSet)
if _, err := config.ValidatePublishLockRules(remoteLocks, cfg.Pipeline.Scriptorium, cfg.Pipeline.Notarius, "locks"); err != nil {
return fmt.Errorf("locks add: %w", err)
}
if err := uploadRemoteLockStore(ctx, store, locks.Key, &config.PublishLockStore{Locks: remoteLocks}); err != nil {
return fmt.Errorf("locks add: %w", err)
}
_, err = fmt.Fprintf(out, "narratio session locks add: locked %s\n", source)
return err
}
// LocksRemove removes one remote lock.
func LocksRemove(ctx context.Context, args []string, out io.Writer) error {
fs := flag.NewFlagSet("locks remove", flag.ContinueOnError)
fs.SetOutput(io.Discard)
var flags commonConfigFlags
addCommonConfigFlags(fs, &flags)
source, err := parseSessionIDAndOnePositionalArg("locks remove", "source id", fs, args, &flags.sessionID)
if err != nil {
return err
}
if strings.TrimSpace(flags.sessionID) == "" {
return fmt.Errorf("locks remove: session_id is required")
}
cfg, store, locks, _, err := loadHelperContext(ctx, flags, true)
if err != nil {
return fmt.Errorf("locks remove: %w", err)
}
if _, err := config.ValidatePublishLockRules([]config.PublishLockRule{{Source: source}}, cfg.Pipeline.Scriptorium, cfg.Pipeline.Notarius, "locks remove"); err != nil {
return fmt.Errorf("locks remove: %w", err)
}
remoteSet := lockSourceSet(locks.Remote)
if _, ok := remoteSet[source]; !ok {
if _, static := lockSourceSet(locks.Static)[source]; static {
return fmt.Errorf("locks remove: source %q is locked by pipeline config and cannot be unlocked remotely", source)
}
return fmt.Errorf("locks remove: remote lock for %q does not exist", source)
}
delete(remoteSet, source)
remoteLocks := lockMapValues(remoteSet)
if err := uploadRemoteLockStore(ctx, store, locks.Key, &config.PublishLockStore{Locks: remoteLocks}); err != nil {
return fmt.Errorf("locks remove: %w", err)
}
_, err = fmt.Fprintf(out, "narratio session locks remove: unlocked %s\n", source)
return err
}
func writeLocks(out io.Writer, cfg *config.Config, locks *effectiveLocks) {
if locks == nil || len(locks.All) == 0 {
fmt.Fprintln(out, "Publish locks: none")
return
}
fmt.Fprintln(out, "Publish locks:")
published := map[string]config.PublishOutputRule{}
if cfg != nil && cfg.Pipeline != nil && cfg.Pipeline.Publish != nil {
for _, rule := range cfg.Pipeline.Publish.Outputs {
published[strings.TrimSpace(rule.Source)] = rule
}
}
staticSet := lockSourceSet(locks.Static)
for _, lock := range locks.All {
origin := "remote"
if _, ok := staticSet[lock.Source]; ok {
origin = "pipeline"
}
promo := "not-published"
if _, ok := published[lock.Source]; ok {
promo = "published"
}
reason := strings.TrimSpace(lock.Reason)
if reason == "" {
reason = "(no reason)"
}
fmt.Fprintf(out, "- %s origin=%s %s reason=%s\n", lock.Source, origin, promo, reason)
}
}
func lockMapValues(in map[string]config.PublishLockRule) []config.PublishLockRule {
keys := make([]string, 0, len(in))
for key := range in {
keys = append(keys, key)
}
sort.Strings(keys)
out := make([]config.PublishLockRule, 0, len(keys))
for _, key := range keys {
item := in[key]
item.Source = key
item.Reason = strings.TrimSpace(item.Reason)
out = append(out, item)
}
return out
}

Some files were not shown because too many files have changed in this diff Show More