52 Commits

Author SHA1 Message Date
c5c35cd3b4 Moved example configuration from docs/examples/ to top-level examples/
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-19 19:46:40 -05:00
574b1cde6c Update documentation for the new analyze stage and artifact registry 2026-05-19 19:42:28 -05:00
ebb21b9201 Removed the legacy built-in session_recap from the analyze stage 2026-05-19 19:26:07 -05:00
958f446387 Add archive stage integration test for the new analyze stage features 2026-05-19 19:12:10 -05:00
86caf4b222 Update analyze-stage metadata and manifest output 2026-05-19 19:07:44 -05:00
e38ed8ba97 Refactor the analyze stage to actually produce the configured artifacts 2026-05-19 18:58:28 -05:00
3e79cf4724 Update artifact resolution so configured artifact IDs are resolved through the runtime catalog 2026-05-19 18:49:27 -05:00
859ae1ae10 Add abstractions for the internal artifact catalog 2026-05-19 18:43:55 -05:00
c63ecbab32 Add new configuration fields and CLI flags for the upcoming analyze stage enhancements 2026-05-19 18:36:30 -05:00
8480b74283 Updated the roadmap for configurable artifact generation 2026-05-19 11:21:28 -05:00
087869f7fa Added a roadmap for new work to support configurable artifacts defined at runtime 2026-05-19 09:26:46 -05:00
2b2a314d65 Move documentation for external integrations into the docs/integrations subfolder 2026-05-19 09:12:40 -05:00
08b0f4edc5 Removed legacy transcript artifact aliases 2026-05-19 09:00:41 -05:00
571a289296 Added new internal documentation 2026-05-19 08:47:45 -05:00
9f80635b42 Updated configuration docs to reflect the minimal pipeline config 2026-05-19 08:15:12 -05:00
11a3e174b6 Set default value for workspace.root and updated config documentation 2026-05-19 07:07:19 -05:00
9c5e5d6dc1 Simplified the reference tables in docs/config.md 2026-05-19 06:51:55 -05:00
c4e87f58c7 Complete documentation rebuild 2026-05-18 22:03:59 -05:00
37daab7857 Bugfix involving nested directory creation
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-18 11:52:40 +00:00
2356688cb9 Removed legacy interfaces and old documentation references to the previous on-disk layout 2026-05-18 03:02:22 +00:00
1054b64d9f Implement minimal downstream invalidation after forced upstream reruns 2026-05-18 01:50:51 +00:00
01fb02426c Update the analyze stage to utilize the new artifact package 2026-05-18 01:29:18 +00:00
7dc79e052f Aligned the archive stage with the new work directory layout 2026-05-18 01:13:51 +00:00
cb525c0f72 Implemented run-local stage execution + immediate promotion for core output-producing stages 2026-05-18 00:55:35 +00:00
622677d038 Added run manifest scaffolding and helpers 2026-05-17 21:15:51 +00:00
550288e008 Add campaign-aware workspace path foundation 2026-05-17 20:57:27 +00:00
e58e545686 Audit workspace architecture implementation plan
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-17 13:22:09 -05:00
6ff54c5a0f Documentation update and reorganization 2026-05-17 13:14:28 -05:00
924b5d15c6 Applied a more general bugfix to path-resolution issues in the archive stage 2026-05-17 11:08:11 -05:00
b065663180 Bugfix involving path resolution in the archive stage 2026-05-17 11:03:02 -05:00
3ba564b00f Bugfix involving directory creation during the merge stage 2026-05-17 08:18:57 -05:00
a3986cf0d6 Centralized defaults into internal/config/defaults.go 2026-05-17 07:53:51 -05:00
539601bd16 Updated the merge stage to normalize the per-speaker transcripts before merging them
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-16 23:30:13 -05:00
6ca1c8d6b0 The backend S3 client now resolves credentials from user-configurable environment variables 2026-05-16 23:22:21 -05:00
4b7b50981b Add .gocache to .gitignore and minor documentation cleanup 2026-05-16 23:21:45 -05:00
33f7ae8f2e Simplify downstream tool configuration
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-16 23:30:40 +00:00
d5a9ad38f8 Add post-archive cleanup policies 2026-05-16 23:09:39 +00:00
6fbefb9867 Add session discovery and template support 2026-05-16 22:57:42 +00:00
1665359486 Added a locally generated UX progress report 2026-05-16 20:11:20 +00:00
03f2543927 Created a UX status report 2026-05-16 20:09:53 +00:00
fe9c348092 Document and review S3 archive workflow 2026-05-16 15:24:44 +00:00
f7f8f1a949 Promote current session artifacts to storage 2026-05-16 15:01:01 +00:00
d40c91acde Upload successful run records to storage 2026-05-16 14:43:29 +00:00
ed4dcf1ef7 Updated go.mod 2026-05-16 09:34:43 -05:00
24cce49a70 Download S3 audio during prepare 2026-05-16 14:33:42 +00:00
1e6db89dd4 Add remote storage backend 2026-05-16 14:22:04 +00:00
0454296c81 Add archive storage path configuration 2026-05-16 14:11:59 +00:00
58c6ab2d54 Updated audita configuration to reflect the new audita public CLI
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-16 08:46:48 -05:00
7995c41675 Update the audita integration documentation reference
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-16 08:02:18 -05:00
62551d43a0 Added filesystem-based secrets loading configuration 2026-05-16 07:56:22 -05:00
8395c12dd3 Update documentation to include an implementation roadmap for the archive stage 2026-05-16 07:36:25 -05:00
dc8e1040f2 Rationalize config file locations and update documentation
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-14 21:02:06 -05:00
125 changed files with 14594 additions and 2422 deletions

5
.gitignore vendored
View File

@@ -2,6 +2,8 @@
.codex
AGENTS.md
.DS_Store
# ---> Go
# If you prefer the allow list template instead of the deny list, see community template:
# https://github.com/github/gitignore/blob/main/community/Golang/Go.AllowList.gitignore
@@ -22,6 +24,9 @@ AGENTS.md
# Dependency directories (remove the comment below to include it)
# vendor/
# Go cache
.gocache
# Go workspace file
go.work
go.work.sum

283
README.md
View File

@@ -1,279 +1,22 @@
# narratio
`narratio` is a Go orchestration application for processing D&D session audio into transcripts and generated artifacts.
Narratio is a Go orchestration application that turns D&D session audio into polished transcripts and generated session artifacts.
## Current Implementation
Implemented now:
- strict config loading/validation (`pipeline.yml` and `session.yml`)
- local workspace/session layout, locking, and manifest persistence
- resumable stage control (`run`, `plan`, `resume`, `run-stage`, `status`)
- real `prepare`, `transcribe`, `merge`, `polish`, `normalize`, `trim`, and `analyze` stages
- real WhisperX, Seriatim, and Audita adapters
- real Scriptorium subprocess adapter
- optional Scriptorium render diagnostics (`render_debug`)
Not implemented yet:
- `archive` stage behavior
- `notify` stage behavior
- additional analyze artifacts beyond `session_recap`
- generic DAG orchestration
## Config Files
Narratio expects two YAML files:
- `pipeline.yml`: pipeline/workspace settings
- `session.yml`: per-session settings
YAML decoding is strict (`KnownFields(true)`), so unknown fields fail fast.
## Canonical Stage Order
1. `prepare`
2. `transcribe`
3. `merge`
4. `polish`
5. `normalize`
6. `trim`
7. `analyze`
8. `archive`
9. `notify`
## Transcript Tiers
- `transcripts/merged.json`: canonical deterministic merged transcript from Seriatim merge
- `transcripts/processed.json`: full raw Audita-polished transcript output
- `transcripts/normalized.json`: Seriatim-normalized transcript from the normalize stage
- `transcripts/trimmed.json`: gameplay-only normalized polished transcript from trim stage
## Normalize Configuration
`pipeline.normalize` is optional. When omitted, Narratio defaults to:
- `output_path: transcripts/normalized.json`
- `output_schema: seriatim-intermediate`
- `report: true`
Allowed `normalize.output_schema` values:
- `seriatim-minimal`
- `seriatim-intermediate`
- `seriatim-full`
`normalize.output_path` is treated as session-workdir-relative when not absolute.
Normalize stage behavior summary:
- normalize runs after `polish` and before `trim`
- normalize resolves `transcripts/processed.json`
- normalize runs Seriatim `normalize` to produce `transcripts/normalized.json`
- normalize diagnostics are written to:
- `artifacts/seriatim.normalize.report.json` (when enabled)
- `logs/seriatim.normalize.stdout.log`
- `logs/seriatim.normalize.stderr.log`
- `config/seriatim.normalize.generated.yml`
## Trim Configuration
`pipeline.trim` is optional. If omitted, no trim config is loaded. If `trim.enabled` is omitted, it defaults to `false`.
When `trim.enabled: true`:
- `trim.output_path` is required
- `trim.bounds.prompt_id` is required
- `trim.bounds.transcript_input_name` is required
- `trim.bounds.output_path` is required
- `trim.bounds.timeout` must be a valid Go duration when provided
- `trim.bounds.render_debug: true` requires `trim.bounds.render_output_path`
- `trim.bounds.profile_id` may be empty to use the prompt default profile
Trim paths are treated as session-workdir-relative when not absolute.
Example trim config:
```yaml
trim:
enabled: true
output_path: "transcripts/trimmed.json"
bounds:
prompt_id: "dnd_session.bounds"
profile_id: ""
transcript_input_name: "transcript"
output_path: "artifacts/session_bounds.json"
timeout: "10m"
render_debug: false
render_output_path: "artifacts/session_bounds.render.json"
seriatim:
report: false
```
Trim behavior summary:
- trim discovers and validates `transcripts/normalized.json`
- trim uses Scriptorium bounds (`dnd_session.bounds` by example config) to produce `artifacts/session_bounds.json`
- bounds IDs are validated against the same normalized transcript ID space that Seriatim trim will consume
- trim converts bounds to Seriatim keep selector (for example `10-868`) and runs Seriatim trim
- if trim is disabled, Narratio copies normalized transcript to trimmed transcript and records `trim_action=copy_disabled`
Trim outputs and diagnostics:
- `artifacts/session_bounds.json`
- `transcripts/trimmed.json`
- `logs/scriptorium.bounds.stdout.log`
- `logs/scriptorium.bounds.stderr.log`
- `config/scriptorium.bounds.generated.yml`
- `logs/seriatim.trim.stdout.log`
- `logs/seriatim.trim.stderr.log`
- `config/seriatim.trim.generated.yml`
- optional bounds render-debug outputs:
- `artifacts/session_bounds.render.json`
- `logs/scriptorium.bounds.render.stdout.log`
- `logs/scriptorium.bounds.render.stderr.log`
- `config/scriptorium.bounds.render.generated.yml`
Render-debug files are diagnostics and are not treated as canonical stage output artifact refs.
## Scriptorium Configuration
`pipeline.scriptorium` is optional. When present, Narratio validates and uses it for analyze-stage artifact generation.
Key points:
- `scriptorium.binary` is required when section is present
- `scriptorium.config_path` is optional
- `scriptorium.timeout` defaults to `10m` when omitted
- `scriptorium.render_debug` enables render diagnostics globally
- artifacts are configured under `scriptorium.artifacts` (map shape supports multiple artifacts)
- enabled artifacts require `prompt_id` and `output_path`
- artifact `render_debug` may override global render setting
- `vars` currently support boolean and string values
Example `session_recap` artifact definition:
```yaml
scriptorium:
binary: "scriptorium"
config_path: "/etc/scriptorium/config.yml"
timeout: "10m"
render_debug: false
artifacts:
session_recap:
enabled: true
prompt_id: "dnd.session_recap"
profile_id: "local-quality" # optional
output_path: "artifacts/session_recap.md"
timeout: "10m"
# render_debug: true # optional per-artifact override
inputs:
transcript:
source: "trimmed_transcript"
required: true
previous_recap:
source: "previous_session_artifact"
artifact: "session_recap"
path: "" # optional; set when available
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: "session_recap"
```
Prompt IDs and profile IDs are configuration values. They are not hardcoded in analyze-stage logic.
Do not put secrets in `pipeline.yml`. If API-key behavior is configured, use env var names only.
## Scriptorium Runtime Behavior
Narratio integrates with Scriptorium through the public CLI subprocess contract:
- generation: `scriptorium run`
- diagnostics/testing: `scriptorium render --format json` when `render_debug` is enabled
For the initial implementation, only `session_recap` generation is supported.
Analyze-stage session recap behavior:
- available transcript input sources for configured artifacts: `processed_transcript`, `normalized_transcript`, `trimmed_transcript`
- session recap should use gameplay-only transcript input (`source: trimmed_transcript`)
- Narratio resolves `trimmed_transcript` from trim manifest output (`transcript_trimmed`) or fallback `transcripts/trimmed.json`
- Narratio resolves `normalized_transcript` from normalize manifest output (`transcript_normalized`) or fallback `transcripts/normalized.json`
- missing trimmed transcript fails clearly and advises running trim stage first
- `normalized_transcript` is the preferred full-transcript source for future table/meta-analysis artifacts
- `processed_transcript` remains supported for advanced/debug use cases
- optionally includes `previous_recap` when configured and resolvable
- omits optional previous recap when unavailable
- fails if required inputs are missing
- validates output file exists and is non-empty
Expected session output paths:
- `artifacts/session_recap.md`
- `logs/scriptorium.session_recap.stdout.log`
- `logs/scriptorium.session_recap.stderr.log`
- `config/scriptorium.session_recap.generated.yml`
- `artifacts/session_recap.render.json` when render diagnostics are enabled
## Examples
Starter files:
- `examples/pipeline.minimal.yml`
- `examples/session.minimal.yml`
- `examples/speakers.yml`
## Commands
Run tests:
It coordinates transcription, merge/polish/normalize/trim processing, artifact generation, archive publishing, and resumable run state in one operator workflow.
```bash
go test ./...
narratio run --session-id 2026-04-04
```
Plan a run:
This command requires discoverable `pipeline.yml` and `session.yml` files (or explicit `--config` and `--session` flags).
```bash
go run ./cmd/narratio plan --config examples/pipeline.minimal.yml --session examples/session.minimal.yml
```
## Documentation
Run full pipeline:
```bash
go run ./cmd/narratio run --config examples/pipeline.minimal.yml --session examples/session.minimal.yml
```
Run analyze only:
```bash
go run ./cmd/narratio run-stage --config examples/pipeline.minimal.yml --session examples/session.minimal.yml analyze
```
## Operational Note
Checksum-based stale detection is not implemented yet.
If prepared inputs or prompt/runtime config change, rerun the appropriate upstream stages before relying on downstream artifacts.
Examples:
- glossary/autocorrect/speaker-context changes: rerun at least `merge`, `polish`, `normalize`, `trim`, and `analyze`
- trim bounds prompt/profile/config changes: rerun at least `normalize`, `trim`, and `analyze`
- session recap prompt/profile/input-source changes: rerun `analyze`
## Roadmap
Near-term roadmap:
- extend analyze to additional configured artifacts
- support workflows where later artifacts consume earlier generated artifacts
- keep orchestration explicit without a generic DAG engine
- implement archive and notify backends
- [Configuration](docs/config.md)
- [CLI Reference](docs/cli.md)
- [Operations and Recovery](docs/operations.md)
- [Troubleshooting](docs/troubleshooting.md)
- [Development Guide](docs/development.md)
- [Architecture Principles](docs/architecture.md)
- [Internal Component Contracts](docs/internal/README.md)
- [Config Examples](examples/)

View File

@@ -1,316 +0,0 @@
# Narratio Architecture
## 1. Purpose
`narratio` is a Go orchestrator for D&D session processing. It runs a stage-based local pipeline from audio input through transcript processing and artifact generation, with manifest-based skip/force/resume behavior.
Narratio integrates with Scriptorium through the **public CLI** (`scriptorium run` and `scriptorium render`) via synchronous subprocess execution.
## 2. Current Status
Implemented:
- strict `pipeline.yml` + `session.yml` loading with strict YAML field checking (`KnownFields(true)`)
- local workspace/session layout, lock file handling, artifact path helpers, checksums, and atomic writes
- manifest store and stage status transitions for resumable runs
- real `prepare`, `transcribe`, `merge`, and `polish` stages
- real WhisperX HTTP adapter
- real Seriatim subprocess adapter
- real Audita subprocess adapter
- real Scriptorium subprocess adapter
- real `normalize` stage producing `transcripts/normalized.json`
- real `trim` stage producing `transcripts/trimmed.json`
- real `analyze` stage for initial `session_recap` generation
- optional Scriptorium render diagnostics (`render_debug`) before production run
Still placeholder/future:
- `archive` stage behavior
- `notify` stage behavior
- additional Scriptorium artifact types beyond `session_recap`
- artifact-to-artifact workflows beyond the initial single-artifact implementation
- generic stale detection based on input/config checksums
## 3. Pipeline and Stage Boundaries
Canonical stage order:
1. `prepare`
2. `transcribe`
3. `merge`
4. `polish`
5. `normalize`
6. `trim`
7. `analyze`
8. `archive`
9. `notify`
Boundary rules:
- orchestration logic lives in `internal/app`
- stage business logic lives in `internal/stage`
- external-tool CLI construction lives in adapter packages
- Scriptorium CLI details stay in `internal/adapters/scriptorium`
## 4. Scriptorium Integration Model
Integration mode:
- public CLI subprocesses only (no Scriptorium internal Go packages, no HTTP API)
- production generation uses `scriptorium run`
- diagnostics/testing render uses `scriptorium render --format json`
Run invocation shape used by adapter:
```bash
scriptorium run --prompt <prompt_id> --input name=path --out <output_path>
```
Optional flags passed when configured:
- `--config <path>`
- `--profile <profile_id>`
- repeated `--var name=value`
- repeated `--input name=path`
- `--timeout <duration>`
- `--api-key-env <ENV_NAME>` when configured
Render invocation shape used by adapter:
```bash
scriptorium render --prompt <prompt_id> --input name=path --format json --out <render_output_path>
```
Adapter behavior:
- always passes `--out`
- captures stdout/stderr separately
- writes generated invocation metadata YAML (redacted, no secrets)
- treats exit code `0` as success
- treats exit code `1` as failure
- treats exit code `2` as failure with `validation_failed=true` and preserves output metadata when available
- validates successful output files exist and are non-empty
- does not treat non-empty stderr as failure by itself
## 5. Configuration Contract
`pipeline.scriptorium` is optional. Existing pipelines without Scriptorium continue to work.
`pipeline.trim` is optional. Existing pipelines without trim config continue to work.
`pipeline.normalize` is optional. Existing pipelines without normalize config continue to work.
When `pipeline.normalize` is omitted, defaults are applied:
- `output_path: transcripts/normalized.json`
- `output_schema: seriatim-intermediate`
- `report: true`
When `pipeline.normalize` is present:
- `output_path` must be non-empty
- `output_schema` must be one of `seriatim-minimal`, `seriatim-intermediate`, or `seriatim-full`
- relative `output_path` values are session-workdir-relative paths
- Seriatim binary settings still come from `pipeline.seriatim`
When `pipeline.trim` is present:
- `enabled` is optional and defaults to `false` when omitted
- relative `output_path`, `bounds.output_path`, and `bounds.render_output_path` values are session-workdir-relative paths
- do not store secrets in trim config values
When `pipeline.trim.enabled: true`:
- `output_path` is required and non-empty
- `bounds.prompt_id` is required and non-empty
- `bounds.transcript_input_name` is required and non-empty
- `bounds.output_path` is required and non-empty
- `bounds.timeout` must parse as a Go duration when provided
- `bounds.render_debug: true` requires non-empty `bounds.render_output_path`
- `bounds.profile_id` may be empty to use the prompt default profile
- prompt IDs are config values, not hardcoded stage logic
When `pipeline.scriptorium` is present:
- `binary` is required and non-empty
- `config_path` is optional; when provided it must be non-empty
- `timeout` is optional; when provided it must parse as a Go duration
- default `timeout` is `10m`
- unknown YAML fields fail strict decode
Artifacts are configured as a map under `pipeline.scriptorium.artifacts` so multiple artifacts are possible in the config shape.
For each artifact definition:
- `enabled: true` requires non-empty `prompt_id`
- `enabled: true` requires non-empty `output_path`
- `timeout` must parse as Go duration when present
- optional per-artifact `render_debug` may override global `scriptorium.render_debug`
- `inputs` are named and each input requires non-empty `source`
- inputs may be optional (`required: false`)
- `vars` values currently support `string` and `bool`
Prompt IDs and profile IDs are configuration values, not hardcoded stage logic.
Trim config shape:
```yaml
trim:
enabled: true
output_path: "transcripts/trimmed.json"
bounds:
prompt_id: "dnd_session.bounds"
profile_id: ""
transcript_input_name: "transcript"
output_path: "artifacts/session_bounds.json"
timeout: "10m"
render_debug: false
render_output_path: "artifacts/session_bounds.render.json"
seriatim:
report: false
```
## 6. Transcript Tiers
Narratio currently produces and uses four transcript tiers:
- `transcripts/merged.json`: canonical deterministic merged transcript from Seriatim merge
- `transcripts/processed.json`: full raw Audita-polished transcript output (includes pre/post-game content)
- `transcripts/normalized.json`: normalized transcript generated by Seriatim normalize
- `transcripts/trimmed.json`: gameplay-only normalized polished transcript from trim stage
Trim reads `transcripts/normalized.json`, validates bounds IDs against that same transcript ID space, and writes `transcripts/trimmed.json`.
## 7. Normalize Stage (Current Implementation)
Normalize stage behavior:
- stage order position: after `polish` and before `trim`
- discovers processed transcript from manifest polish outputs (`transcript_processed`) when present, else `work/<session_id>/transcripts/processed.json`
- validates processed transcript JSON shape (`segments` array required)
- runs Seriatim `normalize` to produce normalized transcript
- validates normalized transcript JSON shape (`segments` array required)
- validates normalize report JSON when enabled
Expected normalize outputs and diagnostics:
- `transcripts/normalized.json`
- `artifacts/seriatim.normalize.report.json` (when normalize report is enabled)
- `logs/seriatim.normalize.stdout.log`
- `logs/seriatim.normalize.stderr.log`
- `config/seriatim.normalize.generated.yml`
## 8. Trim Stage (Current Implementation)
Trim stage behavior:
- stage order position: after `normalize` and before `analyze`
- discovers normalized transcript from manifest normalize outputs (`transcript_normalized`) when present, else `work/<session_id>/transcripts/normalized.json`
- validates normalized transcript JSON shape (`segments` array required)
- when `trim.enabled: false` (or trim config omitted), deterministically copies normalized transcript to `transcripts/trimmed.json` and records `trim_action=copy_disabled`
- when `trim.enabled: true`:
- runs Scriptorium bounds prompt using configured `trim.bounds.prompt_id`
- writes bounds output to configured path (typically `artifacts/session_bounds.json`)
- parses and validates bounds output against the same normalized transcript being trimmed
- converts bounds range to Seriatim keep selector (for example `10-868`)
- runs Seriatim `trim` to produce `transcripts/trimmed.json`
- supports no-trim bounds actions (`none`/`copy`) by copying normalized transcript unchanged
- validates trimmed transcript JSON shape (`segments` array required)
Expected trim outputs and diagnostics:
- `artifacts/session_bounds.json`
- `transcripts/trimmed.json`
- `logs/scriptorium.bounds.stdout.log`
- `logs/scriptorium.bounds.stderr.log`
- `config/scriptorium.bounds.generated.yml`
- `logs/seriatim.trim.stdout.log`
- `logs/seriatim.trim.stderr.log`
- `config/seriatim.trim.generated.yml`
- optional bounds render-debug outputs when enabled:
- `artifacts/session_bounds.render.json`
- `logs/scriptorium.bounds.render.stdout.log`
- `logs/scriptorium.bounds.render.stderr.log`
- `config/scriptorium.bounds.render.generated.yml`
Render-debug files are diagnostics. They are recorded in stage metadata/log/config refs and are not treated as canonical stage output artifact refs.
## 9. Analyze Stage (Current Implementation)
The current real analyze implementation supports only `scriptorium.artifacts.session_recap`.
Behavior:
- if `pipeline.scriptorium` is missing, analyze returns a skipped result with metadata
- if no Scriptorium artifacts are enabled, analyze returns a skipped result with metadata
- if enabled artifacts exist but `session_recap` is not enabled, analyze fails clearly
- available transcript input sources for configured artifacts: `processed_transcript`, `normalized_transcript`, `trimmed_transcript`
- `session_recap` should use `trimmed_transcript` input (`transcripts/trimmed.json`) for in-universe recap generation
- `trimmed_transcript` input is resolved from manifest (`trim` output kind `transcript_trimmed`) when available, otherwise fallback path `work/<session_id>/transcripts/trimmed.json`
- `normalized_transcript` input is resolved from manifest (`normalize` output kind `transcript_normalized`) when available, otherwise fallback path `work/<session_id>/transcripts/normalized.json`
- `processed_transcript` input is resolved from manifest (`polish` output kind `transcript_processed`) when available, otherwise fallback path `work/<session_id>/transcripts/processed.json`
- `normalized_transcript` is the preferred full-transcript source for future table/meta-analysis artifacts
- `processed_transcript` remains available for advanced/debug use cases
- transcript inputs are validated as JSON with top-level `segments` array
- configured inputs are resolved by source
- optional `previous_recap` is omitted when unavailable
- required `previous_recap` fails before invocation when unavailable
- vars are built from config + session metadata
- `render_debug` controls pre-run `scriptorium render` diagnostics
- render failure stops stage before production run
- render output is validated as JSON
- production call uses Scriptorium adapter `RunArtifact`
- successful run output must exist and be non-empty
- missing `trimmed_transcript` input for configured `trimmed_transcript` source fails clearly with guidance to run trim stage first
- manifest records output refs, logs, generated config paths, and non-secret provenance metadata
## 10. Session Recap Paths
Current expected paths for `session_recap`:
- artifact output: `artifacts/session_recap.md`
- run stdout log: `logs/scriptorium.session_recap.stdout.log`
- run stderr log: `logs/scriptorium.session_recap.stderr.log`
- run generated invocation/config: `config/scriptorium.session_recap.generated.yml`
- render output (when enabled): `artifacts/session_recap.render.json`
- render stdout log: `logs/scriptorium.session_recap.render.stdout.log`
- render stderr log: `logs/scriptorium.session_recap.render.stderr.log`
- render generated invocation/config: `config/scriptorium.session_recap.render.generated.yml`
## 11. Security and Privacy
- do not store secrets in pipeline YAML, generated invocation YAML, logs, or manifest metadata
- if API-key integration is configured, pass env var names only (never raw key values)
- avoid logging transcript content or rendered prompt content by default
- treat generated artifacts and logs as potentially sensitive session material
## 12. Operational Caveat (Pre-Stale-Detection)
Checksum-based stale detection is not implemented yet.
If prepared inputs or prompt/runtime configuration change (for example glossary files, prompt IDs, profile IDs, or relevant pipeline settings), rerun the appropriate prior stages to refresh downstream artifacts.
Examples:
- glossary or autocorrect changes usually require rerunning at least `merge`, `polish`, `normalize`, `trim`, and `analyze`
- trim prompt/profile changes require rerunning at least `normalize`, `trim`, and `analyze`
- session recap prompt/profile/input-source changes require rerunning `analyze`
## 13. Roadmap
Planned next steps:
- extend analyze beyond `session_recap` to additional configured artifacts
- support artifact inputs that consume prior generated artifacts
- keep this composable without adding a generic DAG engine in the near term
- implement real `archive` backend behavior
- implement real `notify` backend behavior
- add checksum-based stale detection and stale transitions
Architectural invariants remain:
- strict config decoding/validation
- manifest-driven run control
- clear stage/adapter separation
- configuration-driven prompt/profile/input/vars/output mapping
- Scriptorium integration through public CLI subprocess contract

202
docs/architecture.md Normal file
View File

@@ -0,0 +1,202 @@
# Narratio Architecture
## Purpose
`narratio` is a Go orchestration application for processing D&D session audio into polished transcripts and generated session artifacts.
This document defines the development principles for the project. It is inward-facing: its audience is developers and LLM coding agents. It should guide future changes, not serve as a complete implementation reference.
Implemented component details belong under `docs/internal/`.
## Project Shape
Narratio is a modular, stage-driven orchestrator.
It coordinates specialized downstream systems rather than reimplementing their domains:
- WhisperX handles transcription.
- Seriatim handles deterministic transcript merge/normalization/trim behavior.
- Audita handles transcript correction and polishing.
- Scriptorium handles prompt execution and generated artifacts.
Narratio owns orchestration, configuration loading, session/run state, local and remote path modeling, manifest persistence, stage sequencing, resume behavior, and archive semantics.
Narratio should remain explicit and comprehensible. It is not intended to become a generic workflow engine.
## Core Principles
### Modular and composable
Code should be organized around clear responsibilities. Stages, adapters, config loading, manifest persistence, path construction, and storage behavior should remain separable and independently testable.
### Hexagonal boundaries
External systems should be isolated behind narrow adapters. Stage logic should depend on Narratio-level interfaces and data structures, not on external SDK types, subprocess argument construction, or transport-specific details.
### Standard library preference
Prefer the Go standard library. Add dependencies only when they provide substantial value, are necessary for an external integration, or are a widely used de facto standard.
Accepted examples include a YAML library for configuration and the AWS SDK for S3-compatible storage.
### Explicit orchestration
The pipeline should remain stage-driven and explicit. New behavior should be added through clear stage, adapter, config, or manifest contracts rather than implicit side effects or generic workflow abstraction.
## Stage Design
Each stage should have a clear scope of responsibility.
A stage should define:
- its purpose;
- required input state;
- produced output state;
- config fields it consumes;
- external adapters it uses;
- manifest refs it reads or writes;
- skip, force, and resume behavior;
- failure behavior;
- tests that protect its contract.
Stages should avoid reaching across boundaries. If shared behavior is needed, prefer a helper or service with a narrow interface over duplicating ad hoc logic between stages.
## Transactionality and Resume
A stage should behave transactionally.
A stage is complete only when its outputs have been written, validated, and recorded in the manifest. If a stage fails, Narratio should preserve enough local state for inspection, recovery, and resume.
A failed or incomplete run must not be treated as successful. Later stages should depend on manifest-recorded success, not merely on incidental files existing on disk.
## Manifest Model
The manifest is the durable local ledger for a run.
It should record:
- run identity;
- stage status;
- input and output refs;
- logs and generated config refs;
- checksums or provenance where useful;
- non-secret adapter and archive metadata.
Resume behavior should be manifest-driven. Filesystem state may be inspected and validated, but it should not replace manifest stage state as the source of run progress.
## Adapter Boundaries
Adapters own external integration details.
Expected boundaries:
- WhisperX HTTP details stay in the WhisperX adapter.
- Seriatim CLI construction stays in the Seriatim adapter.
- Audita CLI construction stays in the Audita adapter.
- Scriptorium CLI construction stays in the Scriptorium adapter.
- Object-storage details stay behind the storage adapter interface.
- AWS SDK types stay inside the S3 storage implementation.
Stage code should express intent in Narratio terms and call adapters through narrow contracts.
## Configuration Philosophy
Configuration should be strict, explicit, and operator-friendly.
Principles:
- YAML decoding should reject unknown fields.
- Defaults should be centralized and testable.
- Empty configured values should not silently override meaningful defaults.
- Session templating should remain narrow and deterministic.
- Template support should serve operator convenience, not become a general configuration language.
Narratio should not become a secondary configuration system for downstream tools. Seriatim, Audita, and Scriptorium should own their runtime defaults wherever practical. Narratio should pass required stage-contract paths and explicit operator overrides.
## Path and Storage Discipline
Local and remote paths are part of Narratios application contract.
Code should use centralized path helpers for workspace, spool, session, run, artifact, log, config, and archive paths. Stages should avoid reconstructing canonical paths through scattered string concatenation.
Storage backends should receive explicit bucket-relative keys. Storage implementations should not infer campaign, session, run, or root-prefix semantics.
## Archive Invariants
Archive behavior must preserve a clear commit boundary.
A remote run is current only after the archive stage has successfully uploaded the run record, required promoted outputs, `current/manifest.json`, and finally `current/run_id.txt`.
`current/run_id.txt` is the final remote commit marker and must be written last.
Failed, incomplete, skipped, or uncommitted archive attempts must not be presented as current remote state. Local cleanup is permitted only after successful archive commit and only when explicitly configured.
## Security and Privacy
Narratio handles private campaign material.
Rules:
- Do not store raw secrets in pipeline or session YAML.
- Use environment variable names or secret-file references for secret handling.
- Do not write raw secret values to manifests, logs, generated configs, or archive metadata.
- Treat transcripts, generated artifacts, prompts, reports, and logs as potentially sensitive.
- Avoid logging transcript or prompt content unless there is a deliberate diagnostic reason.
## Diagnostics
Diagnostics should be durable and discoverable, but distinct from canonical outputs.
Logs, reports, generated invocation/config files, and render-debug files support debugging. Transcript tiers and configured artifacts are pipeline products.
Manifest refs should preserve that distinction.
## Determinism
Where practical, Narratio should prefer deterministic behavior:
- stable local path layout;
- stable remote key layout;
- sorted upload order;
- predictable generated config files;
- repeatable command construction;
- tests that do not depend on live external services.
Run IDs and timestamps may be intentionally variable, but surrounding behavior should remain testable.
## Testing Expectations
Core behavior should be testable without live external services.
Tests should cover:
- config loading, defaults, and validation;
- CLI parsing and command construction;
- path helpers;
- manifest transitions;
- stage success, failure, skip, and resume behavior;
- adapter command construction;
- fake storage behavior;
- archive commit ordering;
- example config validity where practical.
Live S3, WhisperX, LLM, or subprocess integration tests should be explicit integration tests, not required for ordinary unit test runs.
## Documentation Expectations
Documentation must follow `docs/documentation/policy.md`.
Current behavior belongs in user-facing docs and `docs/internal/`. Future, planned, aspirational, experimental, or unimplemented work belongs only under `docs/roadmap/`.
`docs/architecture.md` should remain concise and principle-focused. It should not duplicate the full config reference, CLI reference, operations guide, or internal stage documentation.
## Non-Goals
Narratio is not:
- a generic DAG or workflow engine;
- a replacement configuration layer for Seriatim, Audita, or Scriptorium;
- a storage backend abstraction beyond the needs of this pipeline;
- a place to embed raw secrets;
- a place for stage logic to depend directly on AWS SDK types or downstream tool internals;
- a prompt-authoring system.

222
docs/cli.md Normal file
View File

@@ -0,0 +1,222 @@
# CLI
## Shortest Useful Command
```bash
narratio run --session-id 2026-04-04
```
This command uses default config discovery for `pipeline.yml` and `session.yml`; both files must be discoverable unless you pass explicit `--config` and `--session` paths.
## Command Overview
Implemented commands:
- `run`: execute pipeline stages and persist manifest state.
- `plan`: validate config, prepare workspace layout, and print stage run/skip decisions.
- `resume`: continue from first non-succeeded stage unless forced.
- `status`: read and print stage statuses from an existing manifest.
- `run-stage`: execute exactly one stage.
Unknown commands print usage and exit non-zero.
For config semantics, see [docs/config.md](./config.md). For operator lifecycle and recovery, see [docs/operations.md](./operations.md).
## Complete Flag Reference
### `run`
- `--config <path>`: optional explicit `pipeline.yml` path.
- `--session <path>`: optional explicit `session.yml` path.
- `--session-id <value>`: session template variable value.
- `--force`: force stage execution.
- `--artifacts <names>`: analyze artifact keys to execute (repeatable or comma-separated).
### `plan`
- `--config <path>`
- `--session <path>`
- `--session-id <value>`
- `--force`
### `resume`
- `--config <path>`
- `--session <path>`
- `--session-id <value>`
- `--force`
- `--artifacts <names>`: analyze artifact keys to execute (repeatable or comma-separated).
### `run-stage`
- `--config <path>`
- `--session <path>`
- `--session-id <value>`
- `--force`
- `--artifacts <names>`: analyze artifact keys to execute (repeatable or comma-separated).
- positional `<stage>`: required stage name.
Valid stage names:
- `prepare`
- `transcribe`
- `merge`
- `polish`
- `normalize`
- `trim`
- `analyze`
- `archive`
- `notify`
### `status`
- `--manifest <path>`: required manifest path.
## Command Reference
### `run`
Purpose:
- Execute configured stages in canonical order.
Syntax:
```bash
narratio run [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force] [--artifacts <name[,name...]>]
```
Success output:
- `narratio run: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
Common failure cases:
- missing default config/session paths when flags omitted.
- invalid template/rendered session mismatch.
- unknown/invalid `--artifacts` value.
- `--artifacts` with unknown configured artifact key.
### `plan`
Purpose:
- Validate config, load secrets (if configured), prepare workdir, and print stage run/skip decisions.
Syntax:
```bash
narratio plan [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force]
```
Success output includes:
- `narratio plan: workdir prepared at <path>`
- one line per stage (`<stage>: run|skip`)
- `totals: run=<n> skip=<n>`
Common failure cases:
- same config/session discovery and validation failures as `run`.
- secrets directory read failures when `pipeline.secrets.env_dir` is configured.
### `resume`
Purpose:
- Continue from session-manifest stage status.
Syntax:
```bash
narratio resume [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force] [--artifacts <name[,name...]>]
```
Success output:
- `narratio resume: session <session_id> has no remaining stages`
- or `narratio resume: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
Common failure cases:
- same discovery/template/validation failures as `run`.
- manifest load errors when existing manifest is unreadable.
- invalid or unknown artifact selections.
### `status`
Purpose:
- Inspect one manifest file without executing stages.
Syntax:
```bash
narratio status --manifest <manifest.json>
```
Success output includes:
- `session_id: <id>`
- `updated_at: <timestamp>`
- `stages:` entries (`- <stage>: <status>`)
Common failure cases:
- missing `--manifest`.
- unreadable or invalid manifest path.
### `run-stage`
Purpose:
- Execute exactly one stage.
Syntax:
```bash
narratio run-stage [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force] [--artifacts <name[,name...]>] <stage>
```
Success output:
- `narratio run-stage: stage=<name> executed=<n> skipped=<n> force=<true|false>; manifest=<path>`
`--artifacts` behavior:
- accepted only when `<stage>` is `analyze`.
- names are normalized (trimmed, deduplicated, sorted).
- unknown configured artifact keys fail.
Common failure cases:
- missing stage positional arg.
- unknown stage name.
- using `--artifacts` with any non-`analyze` stage.
## Common Workflows
Default-discovery run:
```bash
narratio run --session-id 2026-04-04
```
Run only selected analyze artifacts:
```bash
narratio run --session-id 2026-04-04 --artifacts session_recap,player_handout
```
Resume with selected analyze artifacts:
```bash
narratio resume --session-id 2026-04-04 --artifacts player_handout
```
Run only analyze stage with selected artifacts:
```bash
narratio run-stage --session-id 2026-04-04 --artifacts player_handout analyze
```
## Diagnostic / Recovery Commands
Inspect stage status:
```bash
narratio status --manifest <manifest.json>
```
Get manifest path from previous output:
- `run`, `resume`, and `run-stage` print `manifest=<path>` on success.
## `--artifacts` and `--force`
- `--artifacts` filters which configured artifacts are executable when analyze runs.
- `--artifacts` does not imply `--force`.
- If analyze is already `succeeded` and `--force` is not set, runner-level skip still applies.

304
docs/config.md Normal file
View File

@@ -0,0 +1,304 @@
# Configuration
## 1. Overview
Narratio loads two YAML files:
- `pipeline.yml`: pipeline-level runtime configuration.
- `session.yml`: per-session metadata and input selection.
These commands load and validate both files before running:
- `narratio run`
- `narratio plan`
- `narratio resume`
- `narratio run-stage`
Behavior:
- strict YAML decode is enabled (`KnownFields(true)`): unknown fields fail.
- session templates render before session YAML decode.
- defaults are applied for optional pipeline fields.
- validation enforces required fields, value formats, and cross-field constraints.
## 2. Config file discovery
Pipeline config lookup for `run`, `plan`, `resume`, and `run-stage`:
- If `--config <path>` is provided, that path is used.
- If omitted, Narratio searches in order:
1. `/usr/local/etc/narratio/pipeline.yml`
2. `/etc/narratio/pipeline.yml`
- First existing file wins.
## 3. Session file discovery and templating
Session config lookup for `run`, `plan`, `resume`, and `run-stage`:
- If `--session <path>` is provided, that path is used.
- If omitted, Narratio searches in order:
1. `./session.yml`
2. `/usr/local/etc/narratio/session.yml`
3. `/etc/narratio/session.yml`
- First existing file wins.
Template behavior:
- Supported placeholders:
- `{{session_id}}`
- `{{ session_id }}`
- `--session-id <value>` supplies the placeholder value.
- unresolved placeholders fail load.
- if rendered `session_id` mismatches `--session-id`, load fails.
## 4. Minimal pipeline config
```yaml
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"
```
Why this is sufficient:
- `whisperx.transcribe_url` is required.
- `workspace.root` defaults to `/var/lib/narratio`.
- optional sections (`seriatim`, `audita`, `archive`, `scriptorium`, `trim`, `normalize`, etc.) receive defaults or stay inactive.
## 5. Minimal session template
```yaml
session_id: "{{ session_id }}"
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./examples/speakers.yml
autocorrect_file: ./examples/autocorrect.yml
glossary_file: ./examples/glossary.yml
```
Usage:
```bash
narratio run --config /path/to/pipeline.yml --session ./session.yml --session-id 2026-05-03
```
## 6. Production-oriented config
```yaml
workspace:
root: /var/lib/narratio/workspace
cleanup_after_archive: true
storage:
backend: s3
s3:
bucket: my-dnd-archive
root_prefix: dnd
region: us-east-1
access_key_id_env: OBJECT_STORAGE_KEY_ID
secret_access_key_env: OBJECT_STORAGE_KEY
spool:
root: /var/spool/narratio
delete_audio_after_archive: true
archive:
enabled: true
upload_run: true
promote_artifacts:
- from: transcripts/trimmed.json
to: transcripts/trimmed.json
required: true
- from: artifacts/session_recap.md
to: artifacts/session_recap.md
required: true
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"
scriptorium:
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
output_path: artifacts/session_recap.md
inputs:
transcript:
source: narratio.transcript.trimmed
required: true
```
Operational notes:
- archive promotion is explicit and path-based via `archive.promote_artifacts`.
- Narratio does not auto-promote all generated analyze artifacts.
## 7. Full pipeline reference
| Path | Type | Required | Default |
| --- | --- | --- | --- |
| `pipeline.workspace.root` | string | No | `/var/lib/narratio` |
| `pipeline.workspace.cleanup_after_archive` | bool | No | `false` |
| `pipeline.secrets.env_dir` | string | Conditional | none |
| `pipeline.storage.backend` | string | No | empty |
| `pipeline.storage.bucket` | string | No | empty |
| `pipeline.storage.prefix` | string | No | empty |
| `pipeline.storage.s3.bucket` | string | Conditional | empty |
| `pipeline.storage.s3.root_prefix` | string | No | `dnd` |
| `pipeline.storage.s3.region` | string | No | empty |
| `pipeline.storage.s3.endpoint` | string | No | empty |
| `pipeline.storage.s3.force_path_style` | bool | No | `false` |
| `pipeline.storage.s3.access_key_id_env` | string | No | `OBJECT_STORAGE_KEY_ID` |
| `pipeline.storage.s3.secret_access_key_env` | string | No | `OBJECT_STORAGE_KEY` |
| `pipeline.spool.root` | string | No | `/var/spool/narratio` |
| `pipeline.spool.delete_audio_after_archive` | bool | No | `false` |
| `pipeline.archive.enabled` | bool | No | `true` |
| `pipeline.archive.upload_run` | bool | No | `true` |
| `pipeline.archive.promote_artifacts[]` | list | No | trimmed + session_recap rules |
| `pipeline.archive.promote_artifacts[].from` | string | Yes (per rule) | none |
| `pipeline.archive.promote_artifacts[].to` | string | Yes (per rule) | none |
| `pipeline.archive.promote_artifacts[].required` | bool | No | `true` |
| `pipeline.whisperx.transcribe_url` | string | Yes | none |
| `pipeline.whisperx.language` | string | No | `en` |
| `pipeline.whisperx.timeout` | duration string | No | `30m` |
| `pipeline.whisperx.retries` | int | No | `3` |
| `pipeline.whisperx.retry_delay` | duration string | No | `2s` |
| `pipeline.whisperx.concurrency` | int | No | `2` |
| `pipeline.seriatim.binary` | string | No | `seriatim` |
| `pipeline.seriatim.timeout` | duration string | No | `10m` |
| `pipeline.seriatim.output_schema` | string | No | `seriatim-intermediate` |
| `pipeline.seriatim.coalesce_gap` | float | No | `3.0` |
| `pipeline.seriatim.report` | bool | No | `true` |
| `pipeline.seriatim.env.overlap_word_run_gap` | float | No | unset |
| `pipeline.seriatim.env.overlap_word_run_reorder_window` | float | No | unset |
| `pipeline.seriatim.env.backchannel_max_duration` | float | No | unset |
| `pipeline.seriatim.env.filler_max_duration` | float | No | unset |
| `pipeline.audita.binary` | string | No | `audita` |
| `pipeline.audita.timeout` | duration string | No | `3h` |
| `pipeline.audita.llm_api_key_env` | string | No | empty |
| `pipeline.audita.modules[]` | list[string] | No | empty |
| `pipeline.audita.base_url` | string | No | empty |
| `pipeline.audita.model` | string | No | empty |
| `pipeline.audita.total_llm_concurrency` | int | No | unset |
| `pipeline.audita.proposal_llm_concurrency` | int | No | unset |
| `pipeline.audita.validation_model` | string | No | empty |
| `pipeline.audita.validation_llm_concurrency` | int | No | unset |
| `pipeline.audita.transcript_description` | string | No | empty |
| `pipeline.audita.config_path` | string | No | empty |
| `pipeline.audita.output_schema` | string | No | empty |
| `pipeline.audita.work_dir_retention` | string | No | empty |
| `pipeline.audita.report` | bool | No | `true` |
| `pipeline.normalize.output_path` | string | No | `transcripts/normalized.json` |
| `pipeline.normalize.output_schema` | string | No | `seriatim-intermediate` |
| `pipeline.normalize.report` | bool | No | `true` |
| `pipeline.trim.enabled` | bool | No | `false` |
| `pipeline.trim.output_path` | string | Conditional | none |
| `pipeline.trim.bounds.prompt_id` | string | Conditional | none |
| `pipeline.trim.bounds.profile_id` | string | No | empty |
| `pipeline.trim.bounds.transcript_input_name` | string | Conditional | none |
| `pipeline.trim.bounds.output_path` | string | Conditional | none |
| `pipeline.trim.bounds.timeout` | duration string | No | `10m` |
| `pipeline.trim.bounds.render_debug` | bool | No | `false` |
| `pipeline.trim.bounds.render_output_path` | string | Conditional | none |
| `pipeline.trim.seriatim.report` | bool | No | `false` |
| `pipeline.scriptorium.binary` | string | No | `scriptorium` |
| `pipeline.scriptorium.config_path` | string | No | empty |
| `pipeline.scriptorium.timeout` | duration string | No | `10m` |
| `pipeline.scriptorium.render_debug` | bool | No | `false` |
| `pipeline.scriptorium.artifacts` | map | No | empty |
| `pipeline.scriptorium.artifacts.<name>.enabled` | bool | No | `false` |
| `pipeline.scriptorium.artifacts.<name>.depends_on[]` | list[string] | No | empty |
| `pipeline.scriptorium.artifacts.<name>.render_debug` | bool | No | unset |
| `pipeline.scriptorium.artifacts.<name>.prompt_id` | string | Conditional | none |
| `pipeline.scriptorium.artifacts.<name>.profile_id` | string | No | empty |
| `pipeline.scriptorium.artifacts.<name>.output_path` | string | Conditional | none |
| `pipeline.scriptorium.artifacts.<name>.timeout` | duration string | No | empty |
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source` | string | Conditional | none |
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.artifact` | string | No | empty |
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.path` | string | No | empty |
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.required` | bool | No | `false` |
| `pipeline.scriptorium.artifacts.<name>.vars.<key>` | map value | No | empty |
| `pipeline.analyzer.binary_path` | string | No | empty |
| `pipeline.analyzer.timeout` | duration string | No | empty |
| `pipeline.analyzer.artifacts.output_dir` | string | No | empty |
| `pipeline.analyzer.artifacts.types[]` | list[string] | No | empty |
| `pipeline.notification.backend` | string | No | empty |
| `pipeline.notification.recipient` | string | No | empty |
| `pipeline.notification.timeout` | duration string | No | empty |
Scriptorium artifact-key and dependency rules:
- artifact keys must match `^[a-z][a-z0-9_]*$`.
- enabled artifacts require `prompt_id` and `output_path`.
- `output_path` must be relative, traversal-safe, and under `artifacts/`.
- configured artifact input sources use `narratio.artifact.<name>`.
- if input source references `narratio.artifact.<name>`, artifact `<name>` must exist and must be listed in `depends_on`.
- every `depends_on` entry must be a configured artifact key.
- self-dependency is rejected.
- enabled dependency cycles are rejected.
- any artifact referenced by `depends_on` or `narratio.artifact.<name>` source must define `output_path` (even if not enabled).
Allowed `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source` values:
- `previous_session_artifact`
- `narratio.transcript.merged`
- `narratio.transcript.polished`
- `narratio.transcript.full`
- `narratio.transcript.trimmed`
- `narratio.bounds.session`
- `narratio.artifact.<configured_artifact_key>`
## 8. Full session reference
| Path | Type | Required | Default |
| --- | --- | --- | --- |
| `session.session_id` | string | Yes | none |
| `session.campaign` | string | Yes | none |
| `session.date` | string | No | empty |
| `session.title` | string | No | empty |
| `session.inputs.audio_dir` | string | Conditional | empty |
| `session.inputs.audio_files[]` | list[string] | Conditional | empty |
| `session.inputs.audio_s3.prefix` | string | Conditional | none |
| `session.inputs.speakers_file` | string | Yes | none |
| `session.inputs.autocorrect_file` | string | Yes | none |
| `session.inputs.glossary_file` | string | Yes | none |
Audio-source rule:
- configure exactly one mode:
- `audio_dir`, or
- `audio_files` (at least one), or
- `audio_s3.prefix`
- `audio_s3` cannot be combined with local audio fields.
## 9. Secrets
Narratio supports filesystem-based secret injection via `pipeline.secrets.env_dir`.
Behavior:
- `env_dir` may be absolute or relative.
- relative `env_dir` resolves from current working directory.
- files with valid env-var names (`[A-Za-z_][A-Za-z0-9_]*`) are loaded.
- values are loaded from file contents with trailing newline trimming.
- existing process env vars are preserved.
- invalid names and subdirectories are skipped.
- missing/unreadable `env_dir` fails command execution.
Guidance:
- do not put secret values directly in YAML.
- configure env var names in config and provide values via env/secrets files.
## 10. Examples
Maintained examples:
- `examples/pipeline.minimal.yml`
- `examples/pipeline.production.yml`
- `examples/pipeline.full.annotated.yml`
- `examples/session.template.yml`
- `examples/session.local-audio.yml`
- `examples/session.s3-audio.yml`
These examples are validated by `internal/config` tests.

92
docs/development.md Normal file
View File

@@ -0,0 +1,92 @@
# Development Guide
## Purpose
Canonical contributor workflow and engineering conventions for implemented Narratio behavior.
## Repository layout
- `cmd/narratio/`: CLI entrypoint.
- `internal/app/`: command handlers, plan/run/resume orchestration, cleanup gates, secrets loading.
- `internal/config/`: strict YAML loading, defaults, and validation.
- `internal/stage/`: stage implementations and stage registry/order.
- `internal/adapters/`: external boundary adapters (WhisperX, Seriatim, Audita, Scriptorium, storage, notify).
- `internal/manifest/`: session/run manifest types and persistence.
- `internal/artifacts/`: canonical local/remote path helpers and local artifact store.
- `docs/`: canonical documentation set.
- `examples/`: maintained config examples used by tests.
## Build and test commands
- Run focused CLI behavior checks:
```bash
go test ./internal/app -run TestExecute -v
```
- Run config example load/validate checks:
```bash
go test ./internal/config -run TestExamplesLoadAndValidate -v
```
- Run full test suite:
```bash
go test ./...
```
## Coding conventions
- Keep orchestration explicit and stage-driven; do not introduce generic workflow/DAG abstractions.
- Keep external-system details inside adapter packages; stages should consume Narratio-level contracts only.
- Use centralized path helpers from `internal/artifacts` rather than ad hoc path concatenation.
- Preserve manifest-driven state transitions (`running`, `succeeded`, `failed`, `skipped`, `stale`) as the source of run progress.
- Keep user/operator docs implementation-accurate; planned work belongs only under `docs/roadmap/`.
For design principles and invariants, see [docs/architecture.md](./architecture.md). For stage/adapter contracts, see [docs/internal/README.md](./internal/README.md).
## Dependency policy
- Prefer Go standard library where practical.
- Add third-party dependencies only when they provide clear value for required behavior.
- Keep dependency additions narrow to the boundary package that needs them.
## Change playbooks
### Add config fields
1. Add fields to config structs in `internal/config`.
2. Set defaults in `internal/config/defaults.go` when appropriate.
3. Add validation rules in `internal/config/validate.go`.
4. Add or update load/validate tests in `internal/config/*_test.go`.
5. Update canonical config docs and examples:
- [docs/config.md](./config.md)
- relevant files under `examples/`
### Add CLI flags or commands
1. Update command parsing and behavior in `internal/app`.
2. Add or update command tests (`TestExecute` and command-specific tests).
3. Update [docs/cli.md](./cli.md) and, if operator workflow changes, [docs/operations.md](./operations.md).
### Add or modify stages/adapters
1. Implement stage behavior in `internal/stage` with clear input/output boundaries.
2. Keep external transport/subprocess details in `internal/adapters`.
3. Preserve manifest and promotion semantics expected by runner and archive logic.
4. Add/update stage and adapter tests.
5. Update internal component contracts in `docs/internal/`.
### Update examples
1. Keep canonical examples only in `examples/`.
2. Ensure examples load and validate through runtime config paths.
3. Update `internal/config/load_validate_test.go` as needed.
4. Update links in `docs/config.md` if example filenames change.
### Update docs and roadmap
1. Keep implemented behavior in canonical docs (`README`, `docs/*.md`, `docs/internal/`).
2. Keep planned/unimplemented behavior only in `docs/roadmap/`.
3. After completing roadmap items, remove or mark them complete in `docs/roadmap/documentation.md`.
4. Run a link/path sweep before finalizing changes.

View File

@@ -0,0 +1,356 @@
# Go Project Documentation Policy
## Purpose
Project documentation must help four audiences:
1. users who need to run the application;
2. administrators/operators who need to configure and operate it;
3. developers who need to understand and change it safely;
4. LLM coding agents that need clear scope, boundaries, and invariants.
Docs should be accurate, concise, task-oriented, and organized by audience. Prefer links to canonical docs over repetition.
## Core Rules
### 1. Keep docs concise
Each document should cover a defined scope and only the essentials for that scope.
Avoid:
- long background explanations;
- repeated reference material;
- implementation detail in user-facing docs;
- aspirational language outside roadmap docs;
- verbose examples where one minimal example is clearer.
### 2. Document only implemented behavior outside roadmap files
Unimplemented, planned, aspirational, experimental, or future work may be described only under:
- `docs/roadmap/`
No other documentation file, including `README.md`, should describe code, features, modules, stages, commands, config fields, or behaviors that do not currently exist.
If a feature is partial, non-roadmap docs may describe only the implemented portion and its current boundary.
### 3. Use canonical homes
Each type of information should have one canonical location.
Canonical homes:
- project purpose and quickstart: `README.md`
- development principles: `docs/architecture.md`
- configuration reference: `docs/config.md`
- CLI reference: `docs/cli.md`
- operations and recovery: `docs/operations.md`
- troubleshooting: `docs/troubleshooting.md`
- implemented internals: `docs/internal/`
- future work: `docs/roadmap/`
- contributor workflow: `docs/development.md`
- copyable examples: `examples/`
Other files should summarize briefly and link to the canonical source.
### 4. Keep examples real
Examples should be valid, maintained, and free of secrets.
Where practical:
- example configs should load successfully;
- example commands should match real CLI syntax;
- important examples should be covered by tests.
## Documentation Profiles
All projects require:
- `README.md`
- `docs/architecture.md`
Additional docs depend on the project.
### Small library
Recommended:
- `docs/development.md`, if contributor conventions are non-obvious
### Simple CLI
Required:
- `docs/cli.md`
Recommended:
- `docs/development.md`
### Config-driven CLI
Required:
- `docs/cli.md`
- `docs/config.md`
Recommended:
- `examples/`
- `docs/development.md`
### Stateful or operator-facing application
Required:
- `docs/cli.md`, if CLI-based
- `docs/config.md`, if config-driven
- `docs/operations.md`
Recommended:
- `docs/troubleshooting.md`
- `examples/`
- `docs/development.md`
### Modular, staged, service-oriented, or orchestration application
Required:
- `docs/cli.md`, if CLI-based
- `docs/config.md`, if config-driven
- `docs/operations.md`
- `docs/internal/`
- `docs/development.md`
Recommended:
- `docs/troubleshooting.md`
- validated examples under `examples/`
## Required Documents
### README.md
**Audience:** users, administrators, operators
The README is the outward-facing project orientation page.
It should include, in order:
1. concise description;
2. elevator pitch;
3. shortest useful command or usage example;
4. links to targeted docs.
The README should be short. It is not a manual.
The “shortest useful command” means the simplest command that performs the projects core use case. (It does not mean `app --help`.)
### docs/architecture.md
**Audience:** developers, LLM coding agents
`docs/architecture.md` is required for every project.
It is an inward-facing development policy document. It should describe how the project is intended to be built and changed.
It should include:
- project shape;
- core design principles;
- package and boundary philosophy;
- state/persistence philosophy, if applicable;
- external integration philosophy, if applicable;
- error-handling and logging principles;
- testing expectations;
- documentation expectations;
- architectural invariants;
- explicit non-goals, if useful.
For small projects, this file may be brief. It may simply state that the project is intentionally narrow, monolithic, and dependency-light.
### docs/config.md
**Audience:** administrators, operators, advanced users
Required for applications with configuration files.
It should include, in order:
1. config file locations and discovery precedence;
2. minimal working config;
3. production-oriented config;
4. full configuration reference;
5. secrets handling, if applicable;
6. links to maintained examples.
The full configuration reference should be canonical.
### docs/cli.md
**Audience:** users, administrators, operators
Required for CLI applications.
It should include, in order:
1. shortest useful command;
2. command overview;
3. complete flag reference;
4. common workflows;
5. diagnostic or recovery commands, if applicable.
Explain when commands are useful, not just their syntax.
### docs/operations.md
**Audience:** administrators, operators
Required for applications that maintain state, support resume behavior, run multiple stages, write durable artifacts, use remote storage, or require recovery procedures.
It should cover:
- normal workflow;
- filesystem layout;
- remote storage layout, if applicable;
- logs and manifests;
- resume/retry behavior;
- cleanup behavior;
- archive/backup behavior;
- safe recovery procedures;
- operational caveats.
### docs/troubleshooting.md
**Audience:** administrators, operators
Recommended once recurring failure modes exist.
Each entry should include:
- symptom;
- likely cause;
- diagnostic command or inspection step;
- safe fix;
- relevant links.
### docs/development.md
**Audience:** developers, LLM coding agents
Required for projects maintained by humans and LLM coding agents.
It should include:
- repository layout;
- build/test commands;
- coding conventions;
- dependency policy;
- how to add config fields;
- how to add CLI flags;
- how to add stages/modules/adapters, if applicable;
- how to update examples;
- documentation update expectations.
### docs/internal/
**Audience:** developers, LLM coding agents
Required for modular, staged, service-oriented, or orchestration projects.
This directory describes implemented internal components. It is not the roadmap.
Use one file per major component where useful.
Each component doc should include:
1. purpose;
2. inputs and outputs;
3. boundaries;
4. config fields used;
5. external adapters used;
6. state or manifest behavior, if applicable;
7. skip/resume behavior, if applicable;
8. failure behavior;
9. tests to inspect before changing;
10. architectural invariants.
### docs/roadmap/
**Audience:** maintainers, developers, LLM coding agents
This is the only place for planned, future, aspirational, experimental, or unimplemented work.
Roadmap docs should clearly distinguish:
- proposed work;
- accepted plans;
- deferred ideas;
- rejected ideas;
- implementation prompts or task breakdowns, if useful.
Roadmap docs should not be confused with current behavior.
### docs/integrations/
**Audience:** developers, LLM coding agents
Required for projects that depend on external CLIs, APIs, services, protocols, or file formats where the integration contract is important to maintain.
This directory contains concise, versioned reference notes for external integration contracts. It should document only the parts of the external system that this project actually uses.
Use one file per integration where useful.
## Examples Directory
Projects with non-trivial configuration or workflows should include `examples/`.
Useful examples include:
- minimal working config;
- production-oriented config;
- full annotated config;
- local development config;
- remote/object-storage config;
- minimal session/input file.
Examples should be valid, maintained, tested when practical, and linked from relevant docs.
## Security and Privacy
Docs and examples must not include:
- real API keys;
- tokens;
- passwords;
- private keys;
- private environment dumps;
- sensitive user data;
- raw private transcripts;
- private infrastructure details unless intentionally public.
Document secret-handling mechanisms, not actual secret values.
## Maintenance Rules
When docs change, verify the affected behavior.
Where practical:
- load example config files in tests;
- test CLI examples or command parser behavior;
- validate documented flags against real flags;
- remove stale references;
- update links after renames;
- keep roadmap content out of non-roadmap docs.
If documentation and code disagree, fix the documentation and/or open a roadmap item; do not leave aspirational behavior in current-behavior docs.
Documentation is complete only when it matches the current code.
## Documentation Change Checklist
Before merging documentation changes, verify:
- README is concise and orientation-focused.
- `docs/architecture.md` describes development principles.
- Future work appears only under `docs/roadmap/`.
- User-facing docs avoid unnecessary internals.
- Developer-facing docs preserve boundaries and invariants.
- Config examples match the schema.
- CLI examples match real commands and flags.
- Defaults appear in the canonical config reference.
- No secrets or private data are included.
- Links are accurate.

View File

@@ -0,0 +1,15 @@
# Integration Documentation Index
## Audience
Developers and LLM coding agents changing Narratio's external integration contracts.
## Scope
Implemented-only reference notes for the external systems Narratio currently integrates with.
## Integration Docs
- `audita.md`: Audita adapter invocation and validation contract.
- `seriatim.md`: Seriatim normalize/merge/trim adapter contract.
- `scriptorium.md`: Scriptorium run/render adapter contract.
## Canonical Owner
`docs/integrations/` is the canonical home for external integration reference notes per `docs/documentation/policy.md`.

View File

@@ -1,147 +1,66 @@
# Audita
# Integration: audita
Audita is a framework-first transcript correction application. The public `audita` package provides:
## Purpose
Define Narratio's adapter contract for transcript polishing via Audita CLI subprocess execution.
- deterministic transcript normalization
- token-batched module orchestration
- concrete `glossary`, `homophones`, `spoken_word`, and `grammar` modules built on reusable proposal / validator contracts
- structured run reporting and work-dir diagnostics
## Inputs and Outputs
Inputs (`audita.PolishRequest`):
- merged transcript path
- glossary path
- output processed transcript path
- optional report path (required when report enabled)
- work dir
- generated config path
- stdout/stderr log paths
- optional module/model/base URL and concurrency knobs
The previous working implementation has been preserved as `audita_prototype` inside this repository. Its full regression suite lives under `tests/audita_prototype`.
Outputs (`audita.PolishResult`):
- processed transcript path
- optional report path
- generated config path
- stdout/stderr log paths
- exit code, duration, invoked binary
- adapter metadata
## Development
## Boundaries
Owns:
- Deterministic CLI argument construction for `audita process`
- Environment bridging for API credentials
- Invocation config emission
- Output validation for processed transcript and report
This project is set up for `uv`.
Does not own:
- Upstream/downstream stage orchestration
- Credential sourcing policy beyond required env-var presence check
```sh
uv sync --extra dev
uv run pytest
```
## Config Fields Used
Via `pipeline.audita.*` mapped in app/stage wiring:
- `binary`, `timeout`, `llm_api_key_env`, `modules`, `base_url`, `model`
- `transcript_description`, `config_path`, `output_schema`, `work_dir_retention`
- `total_llm_concurrency`, `proposal_llm_concurrency`, `validation_model`, `validation_llm_concurrency`, `report`
## Usage
## External Adapters Used
- Shared subprocess helper (`internal/adapters/subprocess`) to run CLI and capture logs.
Process a transcript with the current framework implementation:
## State and Manifest Behavior
- No direct manifest writes.
- Stage-level metadata records adapter provenance and credential-present signal.
- Generated invocation YAML is written when `GeneratedConfigPath` is provided.
```sh
uv run audita process transcript.json --glossary glossary.yaml --output corrected.json
```
## Skip and Resume Behavior
- Adapter has no skip/resume logic. Stage/runner controls this.
The framework currently runs this default module sequence:
## Failure Behavior
- Constructor validation fails on invalid binary/timeout/schema/concurrency/URL values.
- Run fails on missing required paths, missing required credential env var, subprocess errors, invalid processed JSON shape, or invalid report JSON.
- Failures preserve stdout/stderr paths in returned result metadata.
1. `glossary`
2. `homophones`
3. `glossary`
4. `spoken_word`
5. `grammar`
## Tests to Inspect Before Changing
- `internal/adapters/audita/subprocess_test.go`
- `internal/adapters/audita/fake_test.go`
- `internal/stage/polish_test.go`
Resolved run instance names are auto-numbered for repeats, so the default report pipeline is:
1. `glossary_1`
2. `homophones`
3. `glossary_2`
4. `spoken_word`
5. `grammar`
The default module sequence is fully implemented today:
- `glossary` proposes glossary-supported acoustic corrections
- `homophones` proposes conservative homophone and mistranscription corrections
- `spoken_word` proposes conservative dysfluency cleanup
- `grammar` proposes punctuation, capitalization, and spacing cleanup only
To run a custom module sequence, pass `--modules`:
```sh
uv run audita process transcript.json --glossary glossary.yaml --modules grammar --output corrected.json
```
To also write a structured JSON report:
```sh
uv run audita process transcript.json --glossary glossary.yaml --output corrected.json --report-json report.json
```
From a checked-out repository, you can also use the root launcher:
```sh
./audita process transcript.json --glossary glossary.yaml --output corrected.json
```
For a system-wide command, install the source tree under `/usr/local/src/audita`, sync dependencies there, and symlink the root launcher into your `PATH`:
```sh
cd /usr/local/src/audita
uv sync --extra dev
ln -s /usr/local/src/audita/audita /usr/local/bin/audita
audita process transcript.json --glossary glossary.yaml --output corrected.json
```
Without `--output`, Audita writes the corrected transcript JSON to stdout and progress logs to stderr.
`--report-json` writes a separate machine-readable run report and never mixes report data into stdout.
Useful configuration can be supplied by CLI flag or environment variable. CLI flags take precedence over environment variables. Normal runs now require LLM API credentials, because the `glossary`, `homophones`, `spoken_word`, and `grammar` modules make real LLM calls. `AUDITA_LLM_API_KEY` and `--llm-api-key` are the preferred provider-neutral credential surfaces, while `OPENROUTER_API_KEY` remains supported as a backward-compatible fallback.
| Environment variable | CLI flag | Default | Purpose |
| --- | --- | --- | --- |
| `AUDITA_MODULES` | `--modules` | `glossary,homophones,glossary,spoken_word,grammar` | Comma-separated logical module keys to run; CLI overrides the environment value |
| `AUDITA_LLM_API_KEY` | `--llm-api-key` | unset | Preferred provider-neutral LLM API credential; CLI overrides both environment-key variants |
| `AUDITA_VALIDATION_LLM_API_KEY` | `--validation-llm-api-key` | unset | Validation-phase LLM API credential; defaults to the primary LLM API key |
| `AUDITA_MODEL` | `--model` | `openrouter/google/gemma-4-31b-it` | LLM model name sent to the configured OpenAI-compatible endpoint |
| `AUDITA_VALIDATION_MODEL` | `--validation-model` | unset | Validation-phase LLM model; defaults to `AUDITA_MODEL` |
| `AUDITA_BASE_URL` | `--base-url` | `https://openrouter.ai/api/v1` | OpenAI-compatible API base URL |
| `AUDITA_VALIDATION_BASE_URL` | `--validation-base-url` | unset | Validation-phase OpenAI-compatible API base URL; defaults to `AUDITA_BASE_URL` |
| `AUDITA_LLM_TIMEOUT_SECONDS` | `--llm-timeout-seconds` | `600` | Per-request timeout in seconds for LLM calls to the configured OpenAI-compatible endpoint |
| `AUDITA_VALIDATION_LLM_TIMEOUT_SECONDS` | `--validation-llm-timeout-seconds` | unset | Validation-phase per-request timeout in seconds; defaults to `AUDITA_LLM_TIMEOUT_SECONDS` |
| `AUDITA_VALIDATION_MAX_PROMPT_TOKENS` | `--validation-max-prompt-tokens` | `2048` | Maximum estimated tokens per validation-phase LLM prompt batch |
| `AUDITA_TARGET_SECTIONS` | `--target-sections` | unset | Exact number of contiguous proposal-stage transcript sections; errors if min/max token bounds cannot be satisfied |
| `AUDITA_MAX_RETRIES` | `--max-retries` | `3` | Maximum Instructor retries for structured responses |
| `AUDITA_VALIDATION_MAX_RETRIES` | `--validation-max-retries` | unset | Validation-phase structured-output retries; defaults to `AUDITA_MAX_RETRIES` |
| `AUDITA_VALIDATION_LLM_CONCURRENCY` | `--validation-llm-concurrency` | unset | Validation-phase LLM concurrency; defaults to `AUDITA_LLM_CONCURRENCY` |
| `AUDITA_MAX_SECTION_TOKENS` | `--max-section-tokens` | `8192` | Maximum estimated tokens per proposal-stage transcript section |
| `AUDITA_MIN_SECTION_TOKENS` | `--min-section-tokens` | `2048` | Minimum estimated tokens per proposal-stage transcript section when balancing for concurrency |
| `AUDITA_GLOSSARY_CONFIDENCE_THRESHOLD` | `--glossary-confidence-threshold` | `0.8` | Minimum confidence required for glossary proposals to survive validation |
| `AUDITA_GRAMMAR_CONFIDENCE_THRESHOLD` | `--grammar-confidence-threshold` | `0.8` | Minimum confidence required for grammar proposals to survive validation |
| `AUDITA_HOMOPHONES_CONFIDENCE_THRESHOLD` | `--homophones-confidence-threshold` | `0.8` | Minimum confidence required for homophone proposals to survive validation |
| `AUDITA_SPOKEN_WORD_CONFIDENCE_THRESHOLD` | `--spoken-word-confidence-threshold` | `0.8` | Minimum confidence required for spoken-word proposals to survive validation |
| `AUDITA_NORMALIZE_MAX_SEGMENT_GAP` | `--normalize-max-segment-gap` | `4.0` | Same-speaker gaps eligible for deterministic merging |
| `AUDITA_NORMALIZE_ELLIPSIS_GAP` | `--normalize-ellipsis-gap` | `3.5` | Same-speaker gaps above this value are joined with ` ... ` |
| `AUDITA_NORMALIZE_MAX_SEGMENT_DURATION` | `--normalize-max-segment-duration` | `60.0` | Maximum merged segment duration |
| `AUDITA_NORMALIZE_MAX_SEGMENT_TOKENS` | `--normalize-max-segment-tokens` | `2048` | Maximum merged segment prompt payload size |
| `AUDITA_WORK_DIR` | `--work-dir` | `/tmp/audita` | Per-run scratch diagnostics directory |
| `AUDITA_WORK_DIR_RETENTION` | `--work-dir-retention` | `auto` | Whether to retain the per-run work directory: `auto`, `always`, or `never` |
Set `AUDITA_MODULES=grammar` to run only the grammar module by default, or override it per command with `--modules`.
Validation-phase LLM settings inherit from the primary `AUDITA_*` LLM settings by default. Set any of the `AUDITA_VALIDATION_*` values only when you want LLM-backed validators to use a different model, endpoint, credential, timeout, retry budget, or concurrency level.
OpenRouter remains the default out of the box:
```sh
export AUDITA_LLM_API_KEY=your-openrouter-key
audita process transcript.json --glossary glossary.yaml --output corrected.json
```
You can point Audita at any OpenAI-compatible endpoint by changing `AUDITA_BASE_URL` and, if needed, `AUDITA_MODEL`. For example, a local vLLM server:
```sh
export AUDITA_LLM_API_KEY=local-dev-key
export AUDITA_BASE_URL=http://localhost:8000/v1
export AUDITA_MODEL=meta-llama/Llama-3.1-8B-Instruct
audita process transcript.json --glossary glossary.yaml --output corrected.json
```
Or the actual OpenAI API:
```sh
export AUDITA_LLM_API_KEY=your-openai-key
export AUDITA_BASE_URL=https://api.openai.com/v1
export AUDITA_MODEL=gpt-4.1-mini
audita process transcript.json --glossary glossary.yaml --output corrected.json
```
`AUDITA_WORK_DIR` stores per-run diagnostics while processing. Under the default `AUDITA_WORK_DIR_RETENTION=auto`, clean successful runs are removed, while failed runs and successful runs with final skipped corrections are preserved. Use `always` to keep every run directory and `never` to remove successful run directories even when skips remain.
Failed runs always preserve the run directory and include an authoritative `report.json` alongside normalization and prompt/response diagnostics.
## Prototype Archive
The archived prototype remains importable as `audita_prototype` and is still covered by its original regression suite. This is intentional: the new `audita` package is a framework-oriented rewrite, not a thin wrapper around the old code.
## Architectural Invariants
- Processed output must be valid JSON with top-level `segments` array.
- When report is enabled, report output must be valid JSON.
- If `llm_api_key_env` is configured, credential must be present in environment.

View File

@@ -1,339 +1,64 @@
# Narratio -> Scriptorium CLI Integration
## 1. Purpose
This document defines how Narratio should invoke Scriptorium through the **public CLI**.
This is a **subprocess integration contract**, not an internal Go API contract.
## 2. Assumptions
- `scriptorium` is installed and available on `PATH`.
- Scriptorium is configured with `config.yml`.
- `config.yml` provides `prompt_dir`, `profile_dir`, and `schema_dir` as needed.
- Prompt and profile libraries are already deployed for the environment.
- Narratio provides prepared artifact files (for example polished transcript, glossary, previous recap, campaign notes).
- Initial integration is synchronous subprocess execution.
- Narratio remains the orchestrator.
In normal operation, Narratio does not need to pass `--prompt-dir` and `--profile-dir` if they are supplied by Scriptorium config.
Narratio may pass `--config <PATH>` when it must use a non-default Scriptorium config file.
## 3. Core Commands Narratio May Call
Primary commands for subprocess integration:
- `scriptorium run`
- `scriptorium render`
For production generation, use `scriptorium run`.
`scriptorium render` is for debugging, dry-runs, test assertions, and validating command construction without LLM execution.
Note: `scriptorium serve` and HTTP API exist, but they are not the initial integration path.
## 4. Command Selection Guidance
- Use `run` to generate an output artifact.
- Use `render` to inspect the prepared prompt and effective settings without calling the LLM.
- Use `render --format json` when Narratio/tests need structured prepare output.
## 5. Recommended `run` Invocation Shape
Production shape:
```bash
scriptorium run \
--prompt <prompt_id> \
--input transcript=<processed-transcript-path> \
--out <output-artifact-path>
```
Common optional additions:
- `--config <path>`: use a specific Scriptorium config file.
- `--profile <profile_id>`: override prompt default profile.
- `--var name=value` (repeatable): small metadata values.
- `--input name=path` (repeatable): additional named artifacts.
- `--timeout <duration>`: per-run timeout override.
- Runtime model override flags (`--llm-base-url`, `--model`, etc.) only for exceptional/operator-directed cases.
## 6. Recommended `render` Invocation Shape
Human-readable debug shape:
```bash
scriptorium render \
--prompt <prompt_id> \
--input transcript=<processed-transcript-path> \
--format text
```
Structured debug/test shape:
```bash
scriptorium render \
--prompt <prompt_id> \
--input transcript=<processed-transcript-path> \
--format json \
--out <render-debug-path>
```
`render` does **not** call the LLM, does **not** validate model output, and does **not** perform repair.
## 7. Inputs
- Pass inputs as repeated `--input name=path` flags.
- `name` must match the Prompt Definition input name.
- Prefer absolute paths, or paths relative to a working directory controlled by Narratio.
- Pass Audita output as the primary transcript input.
- Additional inputs may include glossary, previous recap, campaign notes, event logs, final state maps, or other prompt-specific artifacts.
- Scriptorium reads input files directly; Narratio does not need to inline file content for CLI use.
## 8. Variables
Use repeated `--var name=value` for small metadata values.
Typical examples:
- `session_date`
- `session_id`
- `campaign_name`
- `previous_session_id`
- `output_kind`
Large content belongs in input files, not `--var` values.
## 9. Prompt IDs and Output Artifact Types
Narratio should treat prompt IDs as configuration, not hardcoded business logic.
Narratio config may map stage/output names to prompt IDs, for example:
- session recap prompt
- structured event extraction prompt
- glossary suggestion prompt
- player-facing summary prompt
Prompt IDs used by Narratio should come from the deployed Scriptorium prompt library.
## 10. Profiles
- Prompts may declare `default_profile`.
- Narratio may omit `--profile` to use prompt default profile.
- Narratio may pass `--profile` to force profile selection.
- This enables environment/profile selection like `local-fast`, `local-quality`, `frontier`, `batch`, or test profiles.
- Profile names should generally be Narratio configuration values.
## 11. Runtime Overrides
Supported runtime override flags:
- `--llm-base-url`
- `--model`
- `--api-key-env`
- `--temperature`
- `--max-tokens`
- `--top-p`
- `--timeout`
Guidance:
- Keep normal model/runtime settings in Execution Profiles.
- Use runtime overrides only for explicit per-run exceptions, tests, or operator overrides.
- Never pass raw API keys on the command line.
- `--api-key-env` names an environment variable; Narratio must ensure that variable is set in subprocess environment.
## 12. Config Behavior
- Default config path: `/etc/scriptorium/config.yml`.
- `--config <PATH>` overrides default path.
- Missing default config is allowed by Scriptorium.
- If `--config` is provided explicitly, the file must exist and be valid.
- CLI flags override `config.yml`.
- `config.yml` overrides built-in application defaults.
Narratio can either:
- rely on system default config path, or
- carry an explicit config path and pass `--config`.
## 13. Environment Handling
Subprocess environment recommendations:
- Pass through required API-key environment variables referenced by `api_key_env`.
- Do not pass raw API keys as CLI arguments.
- Avoid logging full environment dumps.
- Capture stdout and stderr separately.
- Use a controlled working directory.
- Prefer absolute artifact paths.
## 14. Output Handling
For `scriptorium run`:
- Use `--out` when Narratio needs durable artifact files.
- Without `--out`, artifact content is written to stdout.
- Preferred orchestration pattern: always use `--out`, then treat the file as stage output artifact.
- Capture stderr for diagnostics.
For `scriptorium render`:
- Use `--out` to store render diagnostics.
- Use `--format json` when tests need to inspect selected profile, effective runtime settings, input hashes, prompt hash, and rendered messages.
## 15. Exit Status and Errors
Current CLI behavior (verified from implementation/tests):
- `0`: success.
- `1`: runtime/parse/config/load/render/generation/IO error.
- `2`: run completed but output validation failed (`ValidationFailed`).
Additional details:
- On `run`, output artifact write happens before exit code selection. If validation fails, artifact may still be written and exit code is `2`.
- `stderr` carries both errors and normal run summary output; non-empty stderr alone does not imply failure.
- `render` returns `0` on success and `1` on failures.
Narratio should treat non-zero exit codes as failed stage execution, but may record generated artifact paths if a run exited `2` and output file exists.
## 16. Recommended Narratio Integration Pattern
1. Build CLI args from Narratio stage configuration.
2. Use subprocess context cancellation/timeout.
3. Pass absolute input paths.
4. Pass `--out` to a session-scoped artifact path.
5. Add `--var` metadata values.
6. Optionally add `--config`.
7. Optionally add `--profile`.
8. Ensure required API-key env vars are present.
9. Run subprocess synchronously.
10. Capture stdout/stderr separately.
11. On success, store output artifact path and invocation metadata in stage artifacts.
12. On failure, store exit code and stderr diagnostics in stage status.
## 17. Suggested Narratio Configuration Shape
Illustrative `pipeline.yml` shape:
```yaml
scriptorium:
binary: scriptorium
config_path: /etc/scriptorium/config.yml
timeout: 10m
render_debug: false
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
profile_id: local-quality # optional
output_path: artifacts/session_recap.md
timeout: 10m
render_debug: false # optional artifact override
inputs:
transcript:
source: trimmed_transcript
required: true
previous_recap:
source: previous_session_artifact
artifact: session_recap
path: "" # optional
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: session_recap
```
The key idea: map Narratio artifact names to prompt ID, optional profile, expected inputs, vars, and output destination.
## 18. Testing Strategy for Narratio Integration
- Use `scriptorium render --format json` to verify command construction without LLM calls.
- Use dedicated test prompt/profile libraries for integration tests.
- Use small fixture transcripts.
- Verify missing-input failure behavior.
- Verify prompt `default_profile` behavior.
- Verify explicit `--profile` override behavior.
- Verify `--config` behavior (default and explicit).
- Verify output file creation when `--out` is used.
- Verify stderr capture on failures.
- Avoid real API keys in tests.
## 19. Security and Privacy Notes
- Never pass raw API keys on command line.
- Do not log full rendered prompts by default; transcripts may contain sensitive content.
- Avoid logging prompt content unless explicit debug mode is enabled.
- Treat generated artifacts as potentially sensitive.
- Use session-scoped, access-controlled output paths.
- `api_key_env` names should come from environment management, not embedded secrets.
## 20. Initial D&D Artifact Generation Examples
These are examples only. Use prompt IDs from the deployed prompt library.
Session recap:
```bash
scriptorium run \
--prompt dnd.session_recap \
--input transcript=/work/session-42/transcript.polished.md \
--input glossary=/work/session-42/glossary.yml \
--out /work/session-42/artifacts/session_recap.md
```
Structured events:
```bash
scriptorium run \
--prompt dnd.structured_events \
--input transcript=/work/session-42/transcript.polished.md \
--out /work/session-42/artifacts/structured_events.json
```
Glossary suggestions:
```bash
scriptorium run \
--prompt dnd.glossary_suggestions \
--input transcript=/work/session-42/transcript.polished.md \
--input previous_recap=/work/session-41/artifacts/session_recap.md \
--out /work/session-42/artifacts/glossary_suggestions.md
```
Player-facing summary:
```bash
scriptorium run \
--prompt dnd.player_summary \
--input transcript=/work/session-42/transcript.polished.md \
--input structured_events=/work/session-42/artifacts/structured_events.json \
--out /work/session-42/artifacts/player_summary.md
```
## 21. Non-Goals
Initial Narratio integration should not:
- call Scriptorium internal Go packages
- use HTTP API as the primary path
- expect Scriptorium to read S3 refs directly
- make Scriptorium responsible for Narratio stage state
- make Scriptorium responsible for notification
- require Scriptorium to understand D&D workflow semantics beyond prompt definitions
## 22. Future Extension Notes
Possible later extensions:
- HTTP API integration
- S3 artifact references if Scriptorium adds S3 reader support
- richer render diagnostics and policy controls
- token budgeting/prompt-size checks
- batch execution if Scriptorium later adds batch support
# Integration: scriptorium
## Purpose
Define Narratio's adapter contract for Scriptorium artifact generation and render-debug subprocess invocations.
## Inputs and Outputs
Inputs:
- `RunArtifactRequest`: binary, config path, prompt/profile IDs, input map, vars map, timeout, output path, logs/config paths, optional API env and working dir
- `RenderArtifactRequest`: same core fields for render mode
Outputs (`ArtifactResult`):
- output path
- stdout/stderr log paths
- generated config path
- exit code and duration
- command mode (`run` or `render`)
- prompt/profile provenance
- validation failure signal
- adapter metadata
## Boundaries
Owns:
- Deterministic CLI arg construction for `scriptorium run` and `scriptorium render`
- Common request validation
- Invocation config emission
- Output existence/non-empty checks
- Validation-failure mapping for run exit code 2
Does not own:
- Artifact selection policy (`analyze` stage)
- Bounds semantic validation (`trim` stage)
## Config Fields Used
Via `pipeline.scriptorium.*` and stage-level artifact config:
- `binary`, `config_path`, `timeout`, `render_debug`
- artifact-level `prompt_id`, `profile_id`, `timeout`, `inputs`, `vars`, `output_path`
## External Adapters Used
- Shared subprocess helper (`internal/adapters/subprocess`).
## State and Manifest Behavior
- No direct manifest writes.
- Stage metadata records adapter outputs and command mode.
- Generated invocation YAML is written when requested.
## Skip and Resume Behavior
- Adapter has no skip/resume logic. Stage/runner controls execution.
## Failure Behavior
- Request validation fails for missing binary/prompt/output, invalid timeout, invalid input/var names, or missing required API env var.
- Subprocess errors bubble with command context.
- `run` exit code 2 is treated as `ValidationFailed=true` and surfaced as error by calling stage.
- Successful subprocess still fails if output file is missing/empty.
## Tests to Inspect Before Changing
- `internal/adapters/scriptorium/subprocess_test.go`
- `internal/adapters/scriptorium/fake_test.go`
- `internal/stage/analyze_test.go`
- `internal/stage/trim_test.go`
## Architectural Invariants
- Both modes require explicit timeout > 0.
- Input/var maps are sorted into deterministic CLI argument order.
- Run-mode validation failures are represented explicitly, not silently skipped.

View File

@@ -1,403 +1,60 @@
# seriatim
`seriatim` merges per-speaker WhisperX-style JSON transcripts into a single JSON transcript that preserves speaker identity and chronological order.
The current implementation supports the `merge` command. It reads one or more input JSON files, optionally maps each input file to a canonical speaker using `speakers.yml`, sorts all segments by timestamp, detects and resolves overlaps when word-level timing is available, assigns consecutive numeric `id` values, and writes a merged JSON artifact.
## Usage
Run from source:
```sh
go run ./cmd/seriatim merge \
--input-file samples/raw/2026-04-19-Eric_Rakestraw.json \
--input-file samples/raw/2026-04-19-Mike_Brown.json \
--output-file merged.json
```
Optional report output:
```sh
go run ./cmd/seriatim merge \
--input-file eric.json \
--input-file mike.json \
--output-file merged.json \
--report-file report.json
```
## CLI
```text
seriatim merge [flags]
```
Global flags:
| Flag | Description |
| --- | --- |
| `--help` | Show command help. |
| `--version` | Show application version. Local builds default to `dev`; release builds inject the release version. |
`merge` flags:
| Flag | Required | Default | Description |
| --- | --- | --- | --- |
| `--input-file` | Yes | none | Input transcript JSON file. Repeat once per speaker/input file. |
| `--output-file` | Yes | none | Merged transcript JSON output path. |
| `--report-file` | No | none | Optional report JSON output path. |
| `--speakers` | No | none | Speaker map YAML file. When omitted, input file basenames are used as speaker labels. |
| `--autocorrect` | No | none | Autocorrect rules YAML file. When omitted, the default `autocorrect` module leaves text unchanged. |
| `--input-reader` | No | `json-files` | Input reader module. |
| `--output-modules` | No | `json` | Comma-separated output modules. |
| `--output-schema` | No | `seriatim-intermediate` | JSON output contract. Allowed values are `seriatim-minimal`, `seriatim-intermediate`, and `seriatim-full`. If omitted, the runtime default is used; consumers that depend on a specific shape should set this explicitly. |
| `--preprocessing-modules` | No | `validate-raw,normalize-speakers,trim-text` | Comma-separated preprocessing modules, evaluated in order. |
| `--postprocessing-modules` | No | `detect-overlaps,resolve-overlaps,backchannel,filler,resolve-danglers,coalesce,detect-overlaps,autocorrect,assign-ids,validate-output` | Comma-separated postprocessing modules, evaluated in order. |
| `--coalesce-gap` | No | `3.0` | Maximum same-speaker gap in seconds for `coalesce`; also used as the `resolve-overlaps` context window. Must be a non-negative float. |
Environment variables:
| Environment Variable | Default | Description |
| --- | --- | --- |
| `SERIATIM_OUTPUT_SCHEMA` | `seriatim-intermediate` | Output schema used when `--output-schema` is not explicitly provided. Allowed values are `seriatim-minimal`, `seriatim-intermediate`, and `seriatim-full`. The CLI flag takes precedence. |
| `SERIATIM_OVERLAP_WORD_RUN_GAP` | `1.0` | Maximum gap in seconds between adjacent timed words when `resolve-overlaps` builds word-run replacement segments. Must be a positive float. |
| `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW` | `1.0` | Near-start window in seconds for ordering replacement word runs shortest-first. Must be a positive float. |
| `SERIATIM_BACKCHANNEL_MAX_DURATION` | `2.0` | Maximum duration in seconds for `backchannel` classification. Must be a positive float. |
| `SERIATIM_FILLER_MAX_DURATION` | `1.25` | Maximum duration in seconds for `filler` classification. Must be a positive float. |
## Input JSON Format
Each input file must be valid JSON with a top-level `segments` array. The current parser accepts the WhisperX segment subset needed for merging:
```json
{
"segments": [
{
"start": 1.25,
"end": 3.5,
"text": "Hello there.",
"words": [
{"word": "Hello", "start": 1.25, "end": 1.55, "score": 0.98},
{"word": "there.", "start": 1.7, "end": 2.0}
]
}
]
}
```
Required segment fields:
- `start`: number, must be `>= 0`.
- `end`: number, must be `>= start`.
- `text`: string.
Optional word fields:
- `words`: array of word timing objects.
- `words[].word`: string.
- `words[].start`: optional number, must be `>= 0` when present.
- `words[].end`: optional number, must be `>= start` when present with `start`.
- `words[].score`: optional number.
- `words[].speaker`: optional raw speaker label string.
Word-level timing is preserved internally for overlap resolution. If a word is missing `start` or `end`, seriatim keeps the word text, emits a warning in the optional report, and does not use that word as a timing anchor. Word timing is not emitted in the final JSON artifact.
## Speaker Map Format
`speakers.yml` maps input files to canonical speaker names using ordered substring rules:
This file is optional. If `--speakers` is omitted, `seriatim` uses each input file basename as the segment speaker label.
```yaml
match:
- speaker: "Eric Rakestraw"
match:
- "Eric_Rakestraw"
- "Eric"
- speaker: "Mike Brown"
match:
- "Mike_Brown"
- "mb"
```
For each `--input-file`, `seriatim` takes the file basename and evaluates the rules in order. The first rule with a matching substring wins, and no later rules are evaluated.
For example, this input:
```text
samples/raw/2026-04-19-Eric_Rakestraw.json
```
matches this rule because the basename contains `Eric_Rakestraw`:
```yaml
- speaker: "Eric Rakestraw"
match:
- "Eric_Rakestraw"
```
Important details:
- Matching is against the input file basename, not the full path.
- Matching is case-insensitive.
- Rules are evaluated from first to last.
- Each rule must have a non-empty `speaker`.
- Each rule must have at least one non-empty `match` string.
- Duplicate speaker names are invalid.
- Every input file must match at least one rule or the command fails.
Deprecated old format:
```yaml
inputs:
eric.json:
speaker: "Eric Rakestraw"
```
The old `inputs:` direct mapping format is no longer supported.
## Output JSON Format
`--output-modules json` controls the writer. `--output-schema` controls the JSON contract that writer serializes.
The named schemas are stable public contracts. If a consumer depends on a specific shape, it should request that schema explicitly at runtime. The runtime default selection may change in a future release.
The `seriatim-intermediate` schema is the current default selection when neither `--output-schema` nor `SERIATIM_OUTPUT_SCHEMA` is set. It stays close to the minimal schema, but adds optional `categories` on each segment:
```json
{
"metadata": {
"application": "seriatim",
"version": "dev",
"output_schema": "seriatim-intermediate"
},
"segments": [
{
"id": 1,
"start": 1.25,
"end": 3.5,
"speaker": "Eric Rakestraw",
"text": "Hello there.",
"categories": ["backchannel"]
}
]
}
```
The `seriatim-full` schema uses the full seriatim envelope:
```json
{
"metadata": {
"application": "seriatim",
"version": "dev",
"input_reader": "json-files",
"input_files": ["eric.json", "mike.json"],
"preprocessing_modules": ["validate-raw", "normalize-speakers", "trim-text"],
"postprocessing_modules": ["detect-overlaps", "resolve-overlaps", "backchannel", "filler", "resolve-danglers", "coalesce", "detect-overlaps", "autocorrect", "assign-ids", "validate-output"],
"output_modules": ["json"]
},
"segments": [
{
"id": 1,
"source": "eric.json",
"source_segment_index": 0,
"speaker": "Eric Rakestraw",
"start": 1.25,
"end": 3.5,
"text": "Hello there.",
"overlap_group_id": 1
},
{
"id": 2,
"source": "eric.json",
"source_ref": "word-run:1:1:1",
"derived_from": ["eric.json#0"],
"speaker": "Eric Rakestraw",
"start": 2.0,
"end": 2.5,
"text": "Resolved word run",
"categories": ["backchannel"]
}
],
"overlap_groups": [
{
"id": 1,
"start": 1.25,
"end": 4.0,
"segments": ["eric.json#0", "mike.json#0"],
"speakers": ["Eric Rakestraw", "Mike Brown"],
"class": "unknown",
"resolution": "unresolved"
}
]
}
```
The `seriatim-minimal` schema emits minimal metadata and compact ordered segments:
```json
{
"metadata": {
"application": "seriatim",
"version": "dev",
"output_schema": "seriatim-minimal"
},
"segments": [
{
"id": 1,
"start": 1.25,
"end": 3.5,
"speaker": "Eric Rakestraw",
"text": "Hello there."
}
]
}
```
Minimal output intentionally omits categories, overlap groups, source/provenance fields, and pipeline configuration metadata.
Intermediate output intentionally omits overlap groups and source/provenance fields, but keeps optional `categories` and minimal metadata.
Segments are sorted deterministically by:
```text
(start, end, source, source_segment_index/source_ref, speaker)
```
Final segment IDs are assigned after sorting and start at `1`.
The public Go output contract is available from:
```go
import "gitea.maximumdirect.net/eric/seriatim/schema"
```
The same package embeds machine-readable JSON Schemas in `schema/full-output.schema.json`, `schema/intermediate-output.schema.json`, and `schema/minimal-output.schema.json`. The default `validate-output` postprocessor validates the selected output shape and verifies final segment IDs are present, sequential, and start at `1`.
## Overlap Detection
The default postprocessing pipeline detects overlapping segment groups.
Overlap behavior:
- A strict timing overlap is required: `next.start < current_group_end`.
- Segments that only touch at a boundary are not grouped.
- Groups require at least two distinct speakers.
- Transitive overlaps are grouped together.
- Segments in detected groups receive `overlap_group_id`.
- `overlap_groups[].segments` contains stable references in `source#source_segment_index` format.
- `class` is currently `unknown`.
- `resolution` is `unresolved` until `resolve-overlaps` replaces the group.
## Overlap Resolution
The default postprocessing pipeline runs `detect-overlaps`, then `resolve-overlaps`, then `backchannel`, then `filler`, then `resolve-danglers`, then `coalesce`, then a second `detect-overlaps` pass.
For each detected overlap group, `resolve-overlaps` uses preserved WhisperX word timing to build smaller word-run replacement segments:
- The resolution window expands the detected overlap group by `--coalesce-gap` seconds on both sides.
- Nearby same-speaker context segments are included when they intersect the expanded window and their start or end is within `--coalesce-gap` of the original overlap boundary.
- Once a segment is selected for replacement, all timed words from that segment participate in word-run construction; the window controls segment selection, not per-word clipping.
- Context segments that are part of another detected overlap group are not pulled into the current group.
- Untimed words are included in replacement text in original word order when nearby timed words create a replacement run.
- Untimed words do not affect replacement segment start/end times or word-run gap splitting.
- Words for the same speaker are merged into one run when the gap between adjacent words is no greater than `SERIATIM_OVERLAP_WORD_RUN_GAP`.
- The default word-run gap is `1.0` seconds.
- Set `SERIATIM_OVERLAP_WORD_RUN_GAP` to a positive number of seconds to override the default.
- Near-start replacement word runs are reordered so shorter segments come first when adjacent starts are within `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW`.
- The default word-run reorder window is `1.0` seconds.
- Set `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW` to a positive number of seconds to override the default.
- Replacement segment text is built by joining word text with single spaces.
- Replacement segments include `source_ref` and `derived_from`.
- Replacement segments omit `source_segment_index` because they are derived from one or more original segments.
- Resolved overlap groups are removed before the second detection pass.
- Replacement segments are left without `overlap_group_id` until the second detection pass annotates any remaining overlap.
- If a speaker has no usable word timing in a group, that speaker's original segment is kept.
- If no speakers in a group have usable word timing, the original group and annotations remain unchanged.
## Backchannels
The default pipeline runs `backchannel` before `coalesce`. It tags short acknowledgement segments with:
```json
"categories": ["backchannel"]
```
Backchannel matching is case-insensitive, ignores punctuation for matching and word-count purposes, trims surrounding whitespace, and requires a matching acknowledgement phrase, no more than three whitespace-delimited words, and duration no greater than `SERIATIM_BACKCHANNEL_MAX_DURATION` seconds. The default maximum duration is `2.0` seconds.
## Fillers
The default pipeline runs `filler` after `backchannel` and before `coalesce`. It tags short filler utterances with:
```json
"categories": ["filler"]
```
Filler matching is case-insensitive, ignores punctuation for matching and word-count purposes, trims surrounding whitespace, and requires only filler tokens such as `um`, `uh`, `er`, `erm`, `ah`, `eh`, `hmm`, `mm`, or repeated combinations of those tokens. Matching segments must contain no more than three whitespace-delimited words and have duration no greater than `SERIATIM_FILLER_MAX_DURATION` seconds. The default maximum duration is `1.25` seconds.
## Dangler Resolution
The default pipeline runs `resolve-danglers` before `coalesce` and before the second overlap detection pass. It repairs short derived fragments when they share provenance with a nearby segment:
- Dangling-end fragments have no more than two words and end in punctuation.
- Dangling-start fragments have no more than two words.
- Matching uses same-speaker segments with any shared `derived_from` value.
- Merged segments use `source_ref` values such as `resolve-danglers:1`, keep the target segment's transcript position, and union `derived_from`.
## Coalescing
The default pipeline runs `coalesce` after `resolve-danglers` and before the second overlap detection pass. It merges adjacent same-speaker segments in the transcript's current order when `next.start - current.end <= --coalesce-gap`.
Coalesced segments use `source_ref` values such as `coalesce:1`, include `derived_from`, and omit `source_segment_index`.
Different-speaker backchannel and filler segments do not block coalescing of surrounding same-speaker segments. Same-speaker backchannel and filler segments are merged normally when they are within `--coalesce-gap`. When same-speaker segments are coalesced, any `backchannel` or `filler` category from the merged inputs is dropped from the coalesced segment.
## Autocorrect
Autocorrect is included in the default postprocessing pipeline. If `--autocorrect` is omitted, the module leaves transcript text unchanged and records a skip event in the optional report.
Enable corrections by passing `--autocorrect`:
```sh
go run ./cmd/seriatim merge \
--input-file input.json \
--autocorrect autocorrect.yml \
--output-file merged.json
```
`autocorrect.yml` format:
```yaml
autocorrect:
- target: "Hrank"
match:
- "hrank"
- "Frank"
- target: "Mike Brown"
match:
- "Mike Pat"
```
Matching behavior:
- Matching is case-sensitive.
- Matches apply only to whole tokens, not substrings inside larger words.
- Punctuation and whitespace can surround a match.
- Multi-word and hyphenated matches are supported.
- Duplicate match strings are invalid, including duplicates across separate rules.
## Current Limitations
- Only JSON input is supported.
- Overlap resolution depends on WhisperX word timing; groups without usable word timing remain unresolved.
- Alternate output formats are not implemented yet.
## Release Builds
Local builds record version metadata as `dev`. Release builds should inject the release version with `ldflags`:
```sh
go build -ldflags "-X gitea.maximumdirect.net/eric/seriatim/internal/buildinfo.Version=v1.0.0" ./cmd/seriatim
```
# Integration: seriatim
## Purpose
Define Narratio's adapter contract for merge, normalize, and trim subprocess invocations of Seriatim.
## Inputs and Outputs
Inputs:
- `MergeRequest`: raw/normalized transcript inputs, output path, optional report, speaker/autocorrect paths, logs/config
- `NormalizeRequest`: input transcript, output path, schema, optional report, timeout/log/config
- `TrimRequest`: input transcript, output path, keep selector, timeout/log/config
Outputs:
- `MergeResult`, `NormalizeResult`, `TrimResult` with output paths, logs/config paths, exit code, duration, binary provenance, and metadata.
## Boundaries
Owns:
- Validated deterministic CLI invocation construction
- Optional env tuning propagation for merge
- Invocation config file emission
- JSON output validation
Does not own:
- Transcript input selection/promotion logic (stage-owned)
- Bounds computation (scriptorium/trim-stage-owned)
## Config Fields Used
Via `pipeline.seriatim.*` mapped in app/stage wiring:
- `binary`, `timeout`, `output_schema`, `coalesce_gap`, `report`
- `env.overlap_word_run_gap`
- `env.overlap_word_run_reorder_window`
- `env.backchannel_max_duration`
- `env.filler_max_duration`
## External Adapters Used
- Shared subprocess helper (`internal/adapters/subprocess`).
## State and Manifest Behavior
- No direct manifest writes.
- Stage metadata consumes adapter result fields and preserves generated config/log references.
## Skip and Resume Behavior
- Adapter has no skip/resume logic. Runner controls stage execution.
## Failure Behavior
- Constructor fails for invalid binary/timeout/output-schema/coalesce-gap.
- Merge fails on missing output path/inputs/report path (if enabled), subprocess errors, invalid merged output JSON, invalid report JSON.
- Normalize fails on missing input/output, invalid schema, subprocess errors, invalid normalized output JSON shape, invalid report JSON.
- Trim fails on missing input/output/keep selector, subprocess errors, invalid trimmed output JSON shape.
## Tests to Inspect Before Changing
- `internal/adapters/seriatim/subprocess_test.go`
- `internal/adapters/seriatim/fake_test.go`
- `internal/stage/merge_test.go`
- `internal/stage/normalize_test.go`
- `internal/stage/trim_test.go`
## Architectural Invariants
- Supported output schemas are limited to `seriatim-minimal`, `seriatim-intermediate`, `seriatim-full`.
- Normalize/trim outputs must include `segments` arrays.
- Merge/normalize/trim all route through deterministic subprocess invocation.

28
docs/internal/README.md Normal file
View File

@@ -0,0 +1,28 @@
# Internal Documentation Index
## Audience
Developers and LLM coding agents changing Narratio internals.
## Scope
Implementation-accurate contracts for workspace/state, manifests, stages, artifact resolution, and adapter boundaries.
## Component Docs
- `adapters.md`: external adapter map, runtime wiring, and boundary ownership.
- `storage.md`: remote storage backend contracts and object-store invariants.
- `manifest.md`: session/run manifest schemas, lifecycle transitions, and persistence semantics.
- `artifacts.md`: built-in artifact registry, runtime artifact catalog, and source-resolution behavior.
- `workspace.md`: local state model, manifests, run-local layout, promotion, and cleanup invariants.
- `stage-prepare.md`: input materialization and provenance capture.
- `stage-transcribe.md`: WhisperX transcript generation.
- `stage-merge.md`: Seriatim normalization + merge.
- `stage-polish.md`: Audita transcript polishing.
- `stage-normalize.md`: post-polish normalization.
- `stage-trim.md`: bounds-driven transcript trimming.
- `stage-analyze.md`: dependency-ordered Scriptorium artifact generation for selected configured artifacts.
- `stage-archive.md`: archive upload and current-pointer publish contract.
## External Integration Notes
- `../integrations/README.md`: canonical location for external integration contracts (`audita.md`, `seriatim.md`, `scriptorium.md`).
## Canonical Owner
`docs/internal/` is the canonical home for implemented internals per `docs/documentation/policy.md`.

79
docs/internal/adapters.md Normal file
View File

@@ -0,0 +1,79 @@
# Internal: Adapters
## Purpose
Describe the external adapter boundaries used by Narratio stages and app orchestration, including default runtime wiring.
## Inputs and outputs
Inputs:
- Stage requests passed through adapter interfaces (for example transcription, merge/normalize/trim, polish, artifact generation, object-store operations, notifications).
- Resolved config values used to construct default adapters.
Outputs:
- Adapter-specific result structs (paths, metadata, status/attempt info, duration/exit details).
- Adapter errors returned to stage/app orchestration.
## Boundaries
Owns:
- Transport/process/SDK details at system boundaries (`HTTP`, subprocess CLI invocation, AWS SDK calls).
- Request/response contracts in `internal/adapters/*` packages.
Does not own:
- Stage sequencing, skip/force/resume decisions.
- Manifest transition logic.
- Canonical workspace path policy.
## Config fields used
Default wiring and adapter calls consume:
- `pipeline.whisperx.*`
- `pipeline.seriatim.*`
- `pipeline.audita.*`
- `pipeline.scriptorium.*`
- `pipeline.storage.*` and `pipeline.archive.*` (object-store construction/gating)
- `pipeline.notification.*` (sender boundary exists; placeholder behavior today)
## External adapters used
Runtime env boundary fields (`internal/stage.Env`):
- `whisperx.Client`
- `seriatim.Runner`
- `audita.Runner`
- `scriptorium.Runner`
- `storage.ObjectStore`
- `notify.Sender`
- `analyzer.Runner`
Current execution usage:
- Actively used by implemented stages: `WhisperX`, `Seriatim`, `Audita`, `Scriptorium`, `ObjectStore`, `Notifier`.
- Present but not used by implemented stage set: `Analyzer`, legacy `storage.Backend`.
Default construction in app runner:
- Auto-constructed when not injected: WhisperX HTTP client, Seriatim subprocess runner, Audita subprocess runner, Scriptorium subprocess runner, object store (only when needed), and `notify.NoopSender`.
- Callers can inject test/fake implementations through `app.RunOptions.Env`.
## State and manifest behavior
- Adapters do not directly mutate session/run manifests.
- Stages and runner own manifest writes and stage status transitions.
- Adapter outputs are persisted indirectly through stage result mapping (outputs/logs/generated configs/metadata).
## Skip and resume behavior
- No adapter-level skip/resume semantics.
- Skip/resume/force behavior is decided by app runner using manifest stage state.
## Failure behavior
- Adapter constructors validate config-derived values and fail early on invalid required inputs.
- Adapter run-time failures are returned to stage code with boundary context and are recorded as stage failures by runner logic.
- Subprocess adapters preserve stdout/stderr and generated-config paths to aid diagnosis.
## Tests to inspect before changing
- `internal/adapters/whisperx/http_test.go`
- `internal/adapters/seriatim/subprocess_test.go`
- `internal/adapters/audita/subprocess_test.go`
- `internal/adapters/scriptorium/subprocess_test.go`
- `internal/adapters/storage/*_test.go`
- `internal/adapters/notify/fake_test.go`
- `internal/adapters/analyzer/fake_test.go`
- `internal/app/runner_test.go`
## Architectural invariants
- Stage code depends on adapter interfaces, not transport-specific implementation types.
- External SDK-specific types remain inside adapter implementations.
- Default app wiring must remain deterministic and overrideable via injected env dependencies.

View File

@@ -0,0 +1,87 @@
# Internal: Artifacts
## Purpose
Define Narratio's artifact identity and resolution model for built-in transcript/bounds artifacts and runtime-configured analyze artifacts.
## Inputs and outputs
Inputs:
- artifact sources from config/runtime (`pipeline.scriptorium.artifacts.*.inputs.*.source`)
- session paths and optional session manifest stage outputs
- runtime artifact catalog state for configured artifact sources
Outputs:
- resolved local artifact path and provenance (`ResolvedSessionArtifact`)
- runtime catalog entries for planned/executable/available artifacts
- validation errors for unsupported, missing, or invalid artifact sources
## Boundaries
Owns:
- built-in artifact registry and content validation rules
- runtime artifact catalog for configured artifact source IDs
- source resolution behavior for built-in and configured artifact sources
Does not own:
- artifact generation (stages produce files)
- manifest transition policy
- archive promotion behavior
## Config fields used
- `pipeline.scriptorium.artifacts.<name>.enabled`
- `pipeline.scriptorium.artifacts.<name>.output_path`
- `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source`
## External adapters used
- none
## State and manifest behavior
Built-in registry entries:
| Artifact ID | Canonical file | Producer stage | Output kind |
| --- | --- | --- | --- |
| `narratio.transcript.merged` | `transcripts/merged.json` | `merge` | `transcript_merged` |
| `narratio.transcript.polished` | `transcripts/processed.json` | `polish` | `transcript_processed` |
| `narratio.transcript.full` | `transcripts/normalized.json` | `normalize` | `transcript_normalized` |
| `narratio.transcript.trimmed` | `transcripts/trimmed.json` | `trim` | `transcript_trimmed` |
| `narratio.bounds.session` | `artifacts/session_bounds.json` | `trim` | `session_bounds` |
Runtime catalog entries include built-ins and configured `narratio.artifact.<name>` sources.
Catalog states:
- `planned`: source is registered and known for this run
- `executable`: configured artifact is selected for analyze execution
- `available`: artifact has a usable file path (generated this run or reused from disk)
Resolution behavior:
- built-in sources resolve via manifest producer outputs first, then canonical fallback path
- configured `narratio.artifact.<name>` sources resolve through runtime catalog availability
- configured source lookup requires catalog context
Configured artifact provenance values:
- `generated.current_analyze_run`
- `filesystem.disabled_artifact_output`
Content validation:
- transcript built-ins: JSON with top-level `segments` array
- bounds built-in: valid JSON
- configured artifacts: non-empty text file
## Skip and resume behavior
- resolver and catalog have no direct skip/resume decisions
- stage/runner skip-resume behavior consumes catalog/resolver results
## Failure behavior
- unsupported source -> source validation error
- known source unavailable -> `ErrSessionArtifactNotFound`
- configured source without catalog -> resolution error
- resolved file with invalid content -> validation error
## Tests to inspect before changing
- `internal/artifacts/artifact_resolver_test.go`
- `internal/artifacts/catalog_test.go`
- `internal/stage/analyze_test.go`
- `internal/config/scriptorium_test.go`
## Architectural invariants
- built-in IDs are static and registry-backed
- configured artifact IDs are runtime-derived (`narratio.artifact.<name>`) and catalog-backed
- built-in/source resolution remains deterministic and validation-gated

81
docs/internal/manifest.md Normal file
View File

@@ -0,0 +1,81 @@
# Internal: Manifest
## Purpose
Describe Narratio's durable execution state model for session-level and run-level manifests, including lifecycle transitions and persistence behavior.
## Inputs and outputs
Inputs:
- Session identity and run identity from app orchestration.
- Stage transition events and stage result payloads.
Outputs:
- Session manifest at `{workspace.root}/work/{campaign}/{session_id}/manifest.json`.
- Run manifest at `{workspace.root}/work/{campaign}/{session_id}/runs/{run_id}/manifest.json`.
## Boundaries
Owns:
- Manifest schemas (`Manifest`, `RunManifest`, stage records, error records, input/artifact records).
- Stage status/action transition methods.
- Persistent store contract (`manifest.Store`) and local JSON store implementation.
Does not own:
- Stage implementation details.
- Path construction policy outside manifest file persistence calls.
- CLI command behavior.
## Config fields used
Manifest package itself does not read config directly.
Manifest identity fields are populated by app/stage orchestration from:
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.storage.s3.*` (when archive/S3 identity is set)
## External adapters used
- No external service adapters.
- Uses local filesystem for persistence via `manifest.LocalStore`.
## State and manifest behavior
Session manifest model:
- Tracks durable per-session stage state and provenance (`pending`, `running`, `succeeded`, `failed`, `skipped`, `stale`, `interrupted`).
- Stores resolved inputs, durable artifacts, stage logs/config refs, and stage metadata.
Run manifest model:
- Tracks one invocation (`run_id`) with requested stages and force mode.
- Tracks per-stage action (`run` or `skip`) and per-stage status.
- Tracks overall run status (`running`, `succeeded`, `failed`).
Persistence behavior:
- Load validates required identity/timestamp fields and normalizes maps/records.
- Save updates `updated_at` and writes JSON atomically (temp file + rename).
- Session and run manifests are saved incrementally before/after stage transitions.
Relationship during execution:
- Runner updates both manifests for every stage transition.
- Session manifest is the durable pipeline-progress ledger.
- Run manifest is invocation history and audit record.
- Analyze stage outputs are persisted as `kind=scriptorium_artifact` with `source_id=narratio.artifact.<name>` for configured artifact identity.
## Skip and resume behavior
- Resume and skip decisions are based on session-manifest stage statuses.
- `--force` reruns selected stages and marks downstream succeeded stages as `stale` in session manifest.
- Run manifest records whether each stage was executed or skipped in that invocation.
## Failure behavior
- Stage failure marks both manifests failed for that stage and records error messages/timestamps.
- Save failures are returned immediately and fail the command.
- Invalid/malformed manifest files fail load with explicit validation/decode errors.
## Tests to inspect before changing
- `internal/manifest/manifest_test.go`
- `internal/manifest/run_manifest_test.go`
- `internal/manifest/store_test.go`
- `internal/app/runner_test.go`
- `internal/app/run_control_test.go`
- `internal/app/resume_run_stage_test.go`
## Architectural invariants
- Session manifest is authoritative for stage progression across invocations.
- Run manifest is invocation-scoped and never replaces session manifest as progress authority.
- Manifest writes are atomic and deterministic (JSON + newline, temp rename pattern).

View File

@@ -0,0 +1,84 @@
# Stage: analyze
## Purpose
Execute selected configured Scriptorium artifacts in deterministic dependency order and promote successful outputs to canonical session artifact paths.
## Inputs and Outputs
Inputs:
- configured artifact definitions from `pipeline.scriptorium.artifacts`
- selected artifact filter from runtime (`--artifacts`) when provided
- resolved artifact input sources declared per artifact (`inputs.*.source`)
- optional previous-session file inputs (`previous_session_artifact`)
Outputs:
- one promoted output file per executed configured artifact at that artifact's configured `output_path`
- stage metadata containing generated artifact entries and reused disabled-artifact entries
## Boundaries
Owns:
- runtime artifact catalog construction for analyze execution
- selected-artifact planning and dependency ordering
- per-artifact input resolution, var resolution, timeout/render-debug resolution
- Scriptorium run/render invocation for each selected artifact
- run-local output generation and canonical promotion
Does not own:
- transcript generation/processing stages
- archive promotion policy
- per-artifact resume semantics
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.scriptorium.binary`
- `pipeline.scriptorium.config_path`
- `pipeline.scriptorium.timeout`
- `pipeline.scriptorium.render_debug`
- `pipeline.scriptorium.artifacts.<name>.*`
- `enabled`
- `depends_on`
- `prompt_id`
- `profile_id`
- `timeout`
- `output_path`
- `render_debug`
- `inputs`
- `vars`
## External Adapters Used
- Scriptorium adapter:
- optional `RenderArtifact` (render debug)
- `RunArtifact` (artifact generation)
## State and Manifest Behavior
- If `pipeline.scriptorium` is absent, stage returns success metadata with `skipped=true`.
- If no artifacts are configured, stage returns success metadata with `skipped=true`.
- If zero artifacts are executable after `enabled` + `--artifacts` filtering, stage returns success metadata with `skipped=true`.
- Builds runtime catalog with built-ins and configured artifacts.
- Non-executable configured artifacts are marked available only when their configured output file exists and is valid on disk.
- Executes selected configured artifacts in topological order with deterministic tie-breaking.
- For each generated artifact, records metadata fields including `name`, `source_id`, `output_kind`, `path`, `prompt_id`, `profile_id`, and `provenance`.
- Reused disabled artifacts are recorded separately in `reused_artifacts` with provenance `filesystem.disabled_artifact_output`.
## Skip and Resume Behavior
- Runner-level skip applies when analyze is already `succeeded` and `--force` is not set.
- Analyze remains stage-scoped for resume/skip; there is no per-artifact resume state.
- `--artifacts` filters which configured artifacts are executable when analyze runs; it does not imply `--force`.
## Failure Behavior
- Fails on invalid dependency ordering, unavailable required configured inputs, invalid built-in input prerequisites, render/run adapter failures, validation-failed adapter results, or missing/empty outputs.
- Required configured dependency missing from catalog availability fails clearly before invocation.
- Optional missing inputs are omitted.
## Tests to Inspect Before Changing
- `internal/stage/analyze_test.go`
- `internal/artifacts/catalog_test.go`
- `internal/artifacts/artifact_resolver_test.go`
- `internal/adapters/scriptorium/subprocess_test.go`
## Architectural Invariants
- Configured artifacts are identified by `narratio.artifact.<name>` source IDs.
- Artifact-to-artifact references rely on explicit `depends_on` declarations validated in config.
- Generated analyze outputs are treated uniformly as Scriptorium artifacts.
- Successful outputs must exist and be non-empty before promotion.

View File

@@ -0,0 +1,68 @@
# Stage: archive
## Purpose
Publish run records and promoted session artifacts to object storage, then atomically advance the remote current pointer.
## Inputs and Outputs
Inputs:
- session manifest and prerequisite stage records
- run root contents under `runs/{run_id}/`
- promotion sources from session root (`archive.promote_artifacts`)
Outputs:
- uploaded run files under `{session_prefix}/runs/{run_id}/...`
- uploaded promoted artifacts under `{session_prefix}/...`
- `{session_prefix}/current/manifest.json`
- `{session_prefix}/current/run_id.txt` written last
## Boundaries
Owns:
- Archive enable/disable gate behavior
- Prerequisite stage success enforcement
- Run file collection and upload (excluding `audio/`)
- Promotion rule resolution and upload
- Commit pointer publish order
Does not own:
- Stage execution before archive
- Post-archive local cleanup policy execution (handled by app cleanup logic)
## Config Fields Used
- `pipeline.archive.enabled`
- `pipeline.archive.upload_run`
- `pipeline.archive.promote_artifacts`
- `pipeline.storage.s3.bucket`
- `pipeline.storage.s3.root_prefix`
- `pipeline.workspace.root`
- `session.campaign`
- `session.session_id`
## External Adapters Used
- Object storage backend (`env.ObjectStore`) for upload/list primitives.
## State and Manifest Behavior
- Requires `prepare`, `transcribe`, `merge`, `polish`, `normalize`, `trim`, and `analyze` status `succeeded`.
- Resolves bucket/prefix from manifest identity first, then config fallback.
- Writes metadata including:
- upload counts/paths
- `current_manifest_key`
- `current_run_id_key`
- `current_pointer_written`
- On skipped archive path, returns metadata with `skipped=true` and pointer not written.
## Skip and Resume Behavior
- Stage may self-skip (metadata skip) when archive disabled or run upload disabled.
- Runner-level skip also applies for previously succeeded stage unless forced.
## Failure Behavior
- Fails on missing prerequisite success, missing object store when required, missing run root, missing required promotion source, upload failures, or pointer write failures.
- Pointer semantics are fail-safe: `current/run_id.txt` is not written if prior required uploads fail.
## Tests to Inspect Before Changing
- `internal/stage/archive_test.go`
- `internal/app/post_archive_cleanup_test.go`
## Architectural Invariants
- Run upload excludes `audio/` subtree.
- `current/manifest.json` uploads before `current/run_id.txt`.
- `current/run_id.txt` is the remote publish commit marker.

View File

@@ -0,0 +1,63 @@
# Stage: merge
## Purpose
Normalize per-speaker raw transcripts and merge them into one merged transcript via Seriatim.
## Inputs and Outputs
Inputs:
- `transcripts/raw/*.json`
- `inputs/speakers.yml`
- `inputs/autocorrect.yml`
Outputs:
- `transcripts/merged.json`
- optional `artifacts/seriatim.report.json` (when report enabled)
## Boundaries
Owns:
- Raw transcript discovery/validation
- Per-input normalize calls to Seriatim
- Final merge call to Seriatim
- Run-local log/config/report path wiring
- Promotion of merged/report outputs to canonical paths
Does not own:
- Transcript polishing or downstream artifact generation
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.seriatim.binary`
- `pipeline.seriatim.timeout`
- `pipeline.seriatim.output_schema`
- `pipeline.seriatim.coalesce_gap`
- `pipeline.seriatim.report`
- `pipeline.seriatim.env.*`
## External Adapters Used
- Seriatim adapter:
- `Normalize` for each raw input
- `Run` for final merge
## State and Manifest Behavior
- Reads transcript inputs from transcribe stage outputs in manifest when present; falls back to canonical raw directory.
- Writes run-local outputs/logs/config under `runs/{run_id}/merge/...` when enabled.
- Promotes canonical merged transcript and optional report.
- Records normalized-input provenance and adapter metadata in stage metadata.
## Skip and Resume Behavior
- Runner-level skip applies when already succeeded and not forced.
- Forced rerun of this or upstream stages can stale downstream succeeded stages via runner invalidation.
## Failure Behavior
- Fails on missing/invalid raw transcripts, missing speakers/autocorrect files, normalize failure, merge failure, invalid merged output JSON, or invalid report JSON when enabled.
## Tests to Inspect Before Changing
- `internal/stage/merge_test.go`
- `internal/adapters/seriatim/subprocess_test.go`
## Architectural Invariants
- Merge consumes normalized forms of each raw transcript.
- Merged transcript must validate before promotion.
- Report output is optional and gated by config.

View File

@@ -0,0 +1,56 @@
# Stage: normalize
## Purpose
Normalize the processed transcript into a deterministic intermediate schema for trim and optionally emit a normalize report.
## Inputs and Outputs
Inputs:
- `transcripts/processed.json`
Outputs:
- `transcripts/normalized.json` (or configured normalize output path)
- optional `artifacts/seriatim.normalize.report.json`
## Boundaries
Owns:
- Processed transcript discovery/validation
- Normalize request construction and invocation
- Optional normalize report wiring
- Promotion of normalized transcript and optional report
Does not own:
- Bounds detection or segment trimming
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.normalize.output_path`
- `pipeline.normalize.output_schema`
- `pipeline.normalize.report`
- `pipeline.seriatim.binary`
- `pipeline.seriatim.timeout`
## External Adapters Used
- Seriatim adapter (`Normalize`).
## State and Manifest Behavior
- Reads processed transcript from polish outputs in manifest when present; falls back to canonical path.
- Uses run-local output/report/log/config paths when run layout is enabled.
- Promotes canonical normalized transcript and optional normalize report.
- Records adapter/result metadata including source path selection.
## Skip and Resume Behavior
- Runner-level skip applies when already succeeded and not forced.
- Forced reruns can stale downstream succeeded stages.
## Failure Behavior
- Fails on missing/invalid processed transcript, adapter error, invalid normalized output, or invalid report output when report enabled.
## Tests to Inspect Before Changing
- `internal/stage/normalize_test.go`
- `internal/adapters/seriatim/subprocess_test.go`
## Architectural Invariants
- Normalized output must validate as processed-transcript-compatible JSON (`segments` array required).
- Default normalize config is applied when `pipeline.normalize` is unset.

View File

@@ -0,0 +1,69 @@
# Stage: polish
## Purpose
Polish merged transcript with Audita and produce a processed transcript for downstream normalization/analyze.
## Inputs and Outputs
Inputs:
- `transcripts/merged.json`
- `inputs/glossary.yml`
Outputs:
- `transcripts/processed.json`
- optional `artifacts/audita.report.json` (when report enabled)
## Boundaries
Owns:
- Merged transcript discovery/validation
- Audita invocation request construction
- Run-local logs/config/work-dir/report wiring
- Promotion of processed transcript and optional report
Does not own:
- Upstream merge normalization
- Downstream normalize/trim/analyze logic
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.audita.binary`
- `pipeline.audita.timeout`
- `pipeline.audita.llm_api_key_env`
- `pipeline.audita.modules`
- `pipeline.audita.base_url`
- `pipeline.audita.model`
- `pipeline.audita.transcript_description`
- `pipeline.audita.config_path`
- `pipeline.audita.output_schema`
- `pipeline.audita.work_dir_retention`
- `pipeline.audita.total_llm_concurrency`
- `pipeline.audita.proposal_llm_concurrency`
- `pipeline.audita.validation_model`
- `pipeline.audita.validation_llm_concurrency`
- `pipeline.audita.report`
## External Adapters Used
- Audita adapter (`env.Audita.Run`).
## State and Manifest Behavior
- Reads merged transcript from merge manifest outputs when available; falls back to canonical merged path.
- Uses run-local output/report/log/config/scratch paths when run layout is enabled.
- Promotes canonical `transcripts/processed.json` and optional report.
- Records adapter invocation metadata, credential presence signal, and output provenance in stage metadata.
## Skip and Resume Behavior
- Runner-level skip applies when already succeeded and not forced.
- Forced rerun can stale downstream succeeded stages via runner invalidation.
## Failure Behavior
- Fails on missing/invalid merged transcript, missing glossary, adapter error, invalid processed output shape (`segments` array required), or invalid report JSON when enabled.
## Tests to Inspect Before Changing
- `internal/stage/polish_test.go`
- `internal/adapters/audita/subprocess_test.go`
## Architectural Invariants
- Processed transcript must contain a top-level `segments` array.
- Report behavior is strictly config-gated.
- Stage output canonicalization always ends at `transcripts/processed.json`.

View File

@@ -0,0 +1,74 @@
# Stage: prepare
## Purpose
Materialize all required session inputs into canonical local workspace paths and record input provenance in the session manifest.
## Inputs and Outputs
Inputs:
- `session.yml` (resolved session config)
- `pipeline.resolved.yml` (materialized from resolved pipeline config)
- `speakers.yml`
- `autocorrect.yml`
- `glossary.yml`
- audio source:
- local (`session.inputs.audio_dir` or `session.inputs.audio_files`), or
- S3 (`session.inputs.audio_s3.prefix`)
Outputs:
- `inputs/session.yml`
- `inputs/pipeline.resolved.yml`
- `inputs/speakers.yml`
- `inputs/autocorrect.yml`
- `inputs/glossary.yml`
- `audio/*.flac` in session workdir
- `manifest.Inputs` records with checksums and source metadata
## Boundaries
Owns:
- Input path resolution and validation
- Local copy/materialization of configs and audio files
- S3 audio download to run-scoped spool, then copy into work audio dir
Does not own:
- Transcript generation/processing
- Archive publish behavior
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `session.inputs.speakers_file`
- `session.inputs.autocorrect_file`
- `session.inputs.glossary_file`
- `session.inputs.audio_dir`
- `session.inputs.audio_files`
- `session.inputs.audio_s3.prefix`
- `pipeline.workspace.root`
- `pipeline.spool.root`
- `pipeline.storage.s3.bucket`
- `pipeline.storage.s3.root_prefix`
## External Adapters Used
- Object storage backend (`env.ObjectStore`) for S3 audio list/download when `audio_s3` is configured.
## State and Manifest Behavior
- Ensures workspace layout exists.
- Writes resolved config and input files to canonical `inputs/` paths.
- Records all prepared inputs into `manifest.Inputs` (sorted deterministically by kind/path).
- For S3 audio, records `S3Bucket`, `S3Key`, `S3Size`, `S3ETag`, and `SpoolPath` in each audio input record.
## Skip and Resume Behavior
- Runner-level skip applies when stage already `succeeded` and `--force` is not set.
- Stage itself is deterministic/idempotent for unchanged inputs (`copyFileIfChanged`, `writeBytesIfChanged`).
## Failure Behavior
- Fails on missing required files, invalid audio source combinations, no discoverable `.flac` files, duplicate audio basenames, missing object store for S3 mode, or S3 list/download failures.
## Tests to Inspect Before Changing
- `internal/stage/prepare_test.go`
- `internal/app/session_cli_test.go`
- `internal/config/load_validate_test.go`
## Architectural Invariants
- `audio_dir`/`audio_files` and `audio_s3` are mutually exclusive.
- Audio files must be `.flac`.
- Canonical `inputs/*` and `audio/*` paths are the durable source for downstream stages.

View File

@@ -0,0 +1,58 @@
# Stage: transcribe
## Purpose
Generate per-speaker raw transcripts from prepared audio using WhisperX.
## Inputs and Outputs
Inputs:
- `audio/*.flac` prepared by `prepare`
Outputs:
- `transcripts/raw/<speaker>.json` for each input audio file
## Boundaries
Owns:
- Discovering prepared audio inputs
- Deriving speaker ids from audio basenames
- Parallel WhisperX invocation with bounded concurrency
- Validating produced JSON and promoting run-local outputs
Does not own:
- Transcript merge/polish/normalize/trim/analyze
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.whisperx.transcribe_url`
- `pipeline.whisperx.language`
- `pipeline.whisperx.timeout`
- `pipeline.whisperx.retries`
- `pipeline.whisperx.retry_delay`
- `pipeline.whisperx.concurrency`
## External Adapters Used
- WhisperX adapter (`env.WhisperX.Transcribe`).
## State and Manifest Behavior
- Uses run-local output paths under `runs/{run_id}/transcribe/outputs/...` when run layout is enabled.
- Validates each generated transcript JSON before promotion.
- Promotes canonical outputs to `transcripts/raw/*.json`.
- Records per-file metadata (attempts/status/duration/output path) in stage metadata.
## Skip and Resume Behavior
- Runner-level skip applies for previously succeeded stage unless forced.
- On forced upstream reruns, downstream succeeded stages can be marked `stale` by runner logic.
## Failure Behavior
- Fails if no prepared audio exists, duplicate speaker basenames are detected, adapter output path mismatches expected path, any output JSON is invalid, or one worker fails.
- Cancels in-flight workers after first terminal error.
## Tests to Inspect Before Changing
- `internal/stage/transcribe_test.go`
- `internal/app/whisperx_wiring_test.go`
## Architectural Invariants
- Speaker identity is derived from `.flac` basename and must be unique.
- Every successful speaker output must be valid JSON before promotion.
- Canonical raw transcript set is the only supported merge input surface.

View File

@@ -0,0 +1,75 @@
# Stage: trim
## Purpose
Optionally trim the normalized transcript to session bounds; always produce a durable trimmed transcript.
## Inputs and Outputs
Inputs:
- `transcripts/normalized.json`
Outputs:
- `transcripts/trimmed.json` (or configured trim output path)
- when trim enabled: `artifacts/session_bounds.json`
## Boundaries
Owns:
- Trim-enabled switch behavior
- Bounds generation via Scriptorium artifact run
- Bounds validation against normalized transcript
- Keep-selector derivation and Seriatim trim invocation
- Copy-through behavior when disabled or bounds indicate unchanged transcript
Does not own:
- Upstream normalization
- Downstream artifact analysis
## Config Fields Used
- `session.session_id`
- `session.campaign`
- `pipeline.workspace.root`
- `pipeline.trim.enabled`
- `pipeline.trim.output_path`
- `pipeline.trim.bounds.prompt_id`
- `pipeline.trim.bounds.profile_id`
- `pipeline.trim.bounds.timeout`
- `pipeline.trim.bounds.output_path`
- `pipeline.trim.bounds.transcript_input_name`
- `pipeline.trim.bounds.render_debug`
- `pipeline.trim.bounds.render_output_path`
- `pipeline.seriatim.binary`
- `pipeline.seriatim.timeout`
- `pipeline.scriptorium.binary`
- `pipeline.scriptorium.config_path`
- `pipeline.scriptorium.timeout`
## External Adapters Used
- Scriptorium adapter:
- optional `RenderArtifact` for bounds debug render
- `RunArtifact` for bounds output
- Seriatim adapter:
- `Trim` when bounds indicate trimming is required
## State and Manifest Behavior
- Reads normalized transcript from normalize manifest outputs when available; falls back to canonical path.
- Uses run-local outputs/logs/reports/config/scratch paths when run layout is enabled.
- Promotes canonical trimmed transcript; promotes session bounds when trim enabled.
- Records bounds diagnostics, trim action, keep selector, and adapter metadata.
## Skip and Resume Behavior
- Runner-level skip applies when already succeeded and not forced.
- Forced reruns can stale downstream succeeded stages.
- When `trim.enabled=false`, stage still succeeds by copying normalized to trimmed output.
## Failure Behavior
- Fails on missing/invalid normalized transcript.
- With trim enabled, fails on missing adapters/config, bounds generation/validation errors, invalid bounds JSON, invalid range/segment ids, trim adapter failures, or invalid trimmed output.
## Tests to Inspect Before Changing
- `internal/stage/trim_test.go`
- `internal/adapters/scriptorium/subprocess_test.go`
- `internal/adapters/seriatim/subprocess_test.go`
## Architectural Invariants
- Trim never falls back to processed transcript; normalized transcript is required input.
- `session_bounds` output exists only for enabled trim path.
- Render-debug artifacts are diagnostics and not declared stage outputs.

71
docs/internal/storage.md Normal file
View File

@@ -0,0 +1,71 @@
# Internal: Storage
## Purpose
Document Narratio's remote storage backend contracts and implementations under `internal/adapters/storage`.
## Inputs and outputs
Inputs:
- Resolved storage config (`pipeline.storage.*`).
- Bucket-relative object keys and local file paths from stage/app orchestration.
Outputs:
- Listed/downloaded/uploaded object metadata (`ObjectInfo`).
- Existence checks and storage-layer errors.
## Boundaries
Owns:
- Remote object-store interface and implementation details.
- S3 client wiring and API calls.
- Object key normalization and upload/download/list primitives.
Does not own:
- Session/run prefix semantics.
- Archive commit order semantics.
- Manifest updates.
## Config fields used
- `pipeline.storage.backend`
- `pipeline.storage.s3.bucket`
- `pipeline.storage.s3.region`
- `pipeline.storage.s3.endpoint`
- `pipeline.storage.s3.force_path_style`
- `pipeline.storage.s3.access_key_id_env`
- `pipeline.storage.s3.secret_access_key_env`
## External adapters used
Storage package contracts:
- `ObjectStore` (active remote object-store boundary): `List`, `Download`, `Upload`, `Exists`.
- `Backend` (archive request boundary): currently implemented with `NoopBackend` only.
Implementations:
- `S3Backend`: AWS SDK-backed `ObjectStore` implementation.
- `FakeBackend`: deterministic test `ObjectStore` and archive backend.
- `NoopBackend`: deterministic no-op archive backend for compatibility wiring.
## State and manifest behavior
- Storage implementations are stateless with respect to manifest/session lifecycle.
- Caller supplies fully-qualified bucket-relative keys.
- Storage layer does not infer campaign/session/run/root-prefix semantics.
- Caller controls publish ordering; storage layer executes individual operations in the order invoked.
## Skip and resume behavior
- No storage-level skip/resume behavior.
- Skip/resume decisions are made by stage/app logic before storage calls occur.
## Failure behavior
- `NewObjectStoreFromConfig` fails when no remote backend is configured or required S3 config is missing.
- `S3Backend` constructor fails when required bucket is missing or AWS client setup fails.
- CRUD operations return contextual errors (including not-found behavior via `Exists`).
- Key normalization is applied before operations (`\\` to `/`, leading slash trimmed).
## Tests to inspect before changing
- `internal/adapters/storage/factory_test.go`
- `internal/adapters/storage/s3_backend_test.go`
- `internal/adapters/storage/fake_test.go`
- `internal/adapters/storage/keys_test.go`
- `internal/adapters/storage/archive.go` + consumers in stage tests (`prepare`, `archive`)
## Architectural invariants
- Callers pass full bucket-relative keys.
- Storage backends must not prepend or infer narratio prefixes.
- Remote transport details remain isolated to storage adapter implementations.

View File

@@ -0,0 +1,68 @@
# Workspace internals
## Purpose
Define the local durable and run-local workspace model used by stages, manifests, resume, and archive.
## Inputs and Outputs
Inputs:
- `pipeline.workspace.root`
- `session.campaign`
- `session.session_id`
- generated `run_id`
Outputs:
- Session manifest at `{workspace.root}/work/{campaign}/{session_id}/manifest.json`
- Run manifest at `{workspace.root}/work/{campaign}/{session_id}/runs/{run_id}/manifest.json`
- Canonical durable session directories and run-local stage trees
## Boundaries
Owns:
- Session-level path layout (`inputs/`, `audio/`, `transcripts/`, `artifacts/`, `reports/`, `logs/`, `config/`, `current/`, `runs/`)
- Run-local stage sandbox layout under `runs/{run_id}/{stage}/`
- Session lock acquisition/release (`.lock`)
Does not own:
- Stage business logic
- Remote archive semantics (documented in `stage-archive.md`)
- CLI argument parsing
## Config Fields Used
- `pipeline.workspace.root`
- `pipeline.workspace.cleanup_after_archive`
- `pipeline.spool.root`
- `pipeline.spool.delete_audio_after_archive`
- `session.campaign`
- `session.session_id`
## External Adapters Used
None directly in this subsystem. Stages may use object storage adapters and then write local outputs into this layout.
## State and Manifest Behavior
- Session state is persisted in the session manifest (`manifest.Manifest`).
- Invocation history is persisted per run in run manifests under `runs/{run_id}/manifest.json`.
- During each run, stage outputs are often written run-local first (`runs/{run_id}/{stage}/outputs/...`) and promoted to canonical session paths after stage success.
- `manifest.Artifacts` entries record `ProducerRunID` for durable outputs.
- For S3 audio sessions, `prepare` records spool/work paths and S3 provenance in `manifest.Inputs`.
## Skip and Resume Behavior
- Skip/resume decisions are made in `internal/app` (`run_control.go`, `resume.go`) using stage status in the session manifest.
- `--force` reruns selected stages and marks downstream previously-succeeded stages as `stale`.
- Workspace layout is idempotent (`EnsureLayoutFor`) and reused across runs.
## Failure Behavior
- Failures preserve manifests and run-local files for inspection.
- Lock conflicts fail fast via `ErrLockConflict`.
- Cleanup can fail post-archive; failure is recorded in archive stage metadata and returned by the run.
## Tests to Inspect Before Changing
- `internal/artifacts/local_test.go`
- `internal/stage/run_local_test.go`
- `internal/app/run_control_test.go`
- `internal/app/resume_run_stage_test.go`
- `internal/app/post_archive_cleanup_test.go`
## Architectural Invariants
- Session root is campaign-aware: `{workspace.root}/work/{campaign}/{session_id}`.
- Run roots are always nested: `runs/{run_id}` under the session root.
- Run-local output promotion must end in canonical session paths.
- Cleanup only targets run-scoped directories and must never delete configured root directories.

149
docs/operations.md Normal file
View File

@@ -0,0 +1,149 @@
# Operations
This guide describes the implemented operator lifecycle for Narratio.
For field-level configuration, see [docs/config.md](./config.md). For full command/flag reference, see [docs/cli.md](./cli.md).
## Normal workflow (S3-first path)
1. Upload session `.flac` files to object storage under the session audio prefix.
2. Run Narratio:
```bash
narratio run --session-id 2026-04-04
```
3. Read success output:
- `narratio run: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
- use `manifest=<path>` with `status` for inspection.
Notes:
- default config/session discovery applies unless `--config` and `--session` are passed.
- S3 audio mode requires `session.inputs.audio_s3.prefix` and valid object-store access.
## Local filesystem layout and state artifacts
Session root:
- `{workspace.root}/work/{campaign}/{session_id}/`
Primary state:
- `manifest.json`: session-level stage state.
- `runs/{run_id}/manifest.json`: invocation-level state.
- `.lock`: session lock while a run is active.
Canonical session directories:
- `inputs/`
- `audio/`
- `transcripts/`
- `artifacts/`
- `reports/`
- `logs/`
- `config/`
- `current/`
- `runs/`
Run-local stage directories:
- `runs/{run_id}/{stage}/` with stage-local `outputs/`, `logs/`, `reports/`, `config/`, `scratch/`.
Behavior:
- directory creation is idempotent.
- stage outputs are generally generated run-local first, then promoted to canonical paths on success.
## Analyze artifact execution lifecycle
Analyze executes configured artifacts from `pipeline.scriptorium.artifacts`.
Execution model:
- executable set = enabled artifacts, filtered by `--artifacts` when provided.
- artifact-to-artifact dependencies are declared via `depends_on`.
- selected artifacts run in deterministic dependency order.
- after each successful artifact run, output is promoted to configured canonical `output_path`.
Configured artifact source reuse:
- a non-executable configured artifact can satisfy inputs if its configured output file already exists and is valid.
- reused configured artifact provenance is `filesystem.disabled_artifact_output`.
`--artifacts` behavior:
- accepted on `run`, `resume`, and `run-stage analyze`.
- filters analyze execution only; does not force stage rerun.
## Remote archive layout and publish contract
When archive is enabled and run upload is enabled, archive publishes under:
- session prefix: `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/`
- run prefix: `{session_prefix}/runs/{run_id}/`
Archive uploads:
- run record files from run root (excluding `audio/`).
- promoted files from explicit `archive.promote_artifacts` rules.
Publish order:
1. upload `current/manifest.json`
2. upload `current/run_id.txt` last
`current/run_id.txt` is the remote commit marker.
Archive promotion is explicit and path-based:
- Narratio does not auto-promote all generated analyze artifacts.
- missing required promotion sources fail archive stage.
- missing optional promotion sources are skipped.
## Resume, retry, and safe rerun behavior
Default skip:
- `run` and `run-stage` skip already-succeeded stages unless `--force` is set.
Resume:
- `resume` starts at first non-succeeded stage.
- `resume --force` runs full stage order.
Forced reruns:
- force-rerunning an upstream succeeded stage marks downstream succeeded stages as `stale`.
Safe rerun pattern:
1. rerun the changed stage with `--force`.
2. run `resume` to rebuild downstream stages.
## Cleanup behavior
Cleanup is considered only when archive stage executed and succeeded.
Cleanup toggles:
- `pipeline.spool.delete_audio_after_archive=true` deletes run-scoped spool audio.
- `pipeline.workspace.cleanup_after_archive=true` deletes run-scoped local run directory.
Cleanup eligibility gates:
- archive enabled
- archive run upload enabled
- run record upload completed
- current pointer write completed (`current/run_id.txt` written)
No cleanup for failed/incomplete/unarchived/archive-skipped runs.
## Failure and recovery playbooks
After failure, Narratio keeps:
- session manifest
- run manifest
- run-local artifacts/logs/config/reports
Failed or incomplete runs remain local-only.
Recommended recovery:
1. inspect state:
```bash
narratio status --manifest <manifest-path>
```
2. fix root cause (config/input/credentials/service availability).
3. continue with `resume`, or targeted `run-stage --force` followed by `resume`.
## Operational caveats
- `status` requires explicit `--manifest`; there is no session-id lookup command.
- local and S3 audio input modes are mutually exclusive.
- archive publish requires upstream stages through `analyze` to be `succeeded`.
- required promotion rules can fail when selected analyze artifacts did not generate a required file path.

View File

@@ -0,0 +1,762 @@
# Roadmap: Runtime-Defined Scriptorium Artifacts
## Status
Implementation roadmap for a pre-release hard cutover.
## Purpose
Narratio currently treats artifact generation as a narrow `analyze` stage that supports a hard-coded `session_recap` artifact. This roadmap describes how to generalize artifact generation so operators can define Scriptorium-backed output artifacts at runtime through `pipeline.yml`.
The goal is to keep Narratio as a fixed pipeline orchestrator while making the artifact generation step configurable, composable, deterministic, and easy to regenerate selectively.
## Desired Outcome
Operators should be able to define artifacts such as session recaps, player handouts, NPC summaries, quest logs, entity maps, or other campaign-specific outputs without changing Narratio code.
A configured artifact is declared under:
```text
pipeline.scriptorium.artifacts.<name>
```
Each configured artifact becomes a canonical runtime artifact source ID:
```text
narratio.artifact.<name>
```
For example:
```yaml
scriptorium:
artifacts:
session_recap:
enabled: true
prompt_id: dnd_session.session_recap
output_path: artifacts/session_recap.md
inputs:
transcript:
source: narratio.transcript.trimmed
required: true
```
This artifact is addressable by later artifacts as:
```text
narratio.artifact.session_recap
```
A dependent artifact can then consume it explicitly:
```yaml
scriptorium:
artifacts:
player_handout:
enabled: true
depends_on:
- session_recap
prompt_id: dnd_session.player_handout
output_path: artifacts/player_handout.md
inputs:
recap:
source: narratio.artifact.session_recap
required: true
transcript:
source: narratio.transcript.trimmed
required: true
```
## Resolved Design Decisions
The following decisions are settled for the initial implementation:
1. Configured artifact outputs must live under Narratio's internal artifact output directory, initially `artifacts/`.
2. The artifact output directory should be defined as an internal default in `internal/config/defaults.go`, but no public configuration knob should be exposed yet.
3. Artifact `output_path` should remain explicit in the initial implementation to avoid guessing file extensions or output formats.
4. A disabled artifact may still be referenced as an input if its declared output already exists on disk and passes basic validation.
5. A disabled artifact is not executable during the current analyze run.
6. Artifact-to-artifact references require an explicit `depends_on` entry. Narratio should fail fast if the dependency declaration is missing.
7. The manifest remains stage-oriented: `analyze` succeeds or fails as a full stage.
8. Analyze-stage metadata may record per-artifact output details for provenance and later resolution, but not for intra-stage resume semantics.
9. `--artifacts` should be added as a CLI filter for selective artifact generation.
10. `--artifacts` does not imply `--force`; it only changes which configured artifacts are treated as executable when `analyze` actually runs.
11. Because Narratio is still pre-release, the hard-coded `session_recap` behavior should be removed immediately rather than deprecated gradually.
## Scope
This roadmap covers:
- introducing a runtime artifact catalog;
- generalizing configured Scriptorium artifact execution;
- supporting `narratio.artifact.<name>` source IDs;
- adding explicit artifact dependencies;
- supporting disabled-but-resolvable artifact inputs;
- adding selective artifact execution via `--artifacts`;
- recording generated artifacts in analyze-stage metadata and/or manifest outputs;
- removing hard-coded `session_recap` behavior;
- updating tests and documentation.
## Non-Goals
This feature should not turn Narratio into a general workflow engine.
The initial implementation should not add:
- arbitrary shell-command artifacts;
- arbitrary user-defined stages;
- loops or conditional branching;
- automatic archive promotion of generated artifacts;
- semantic knowledge of particular artifact types;
- per-artifact resume semantics within a successful or failed analyze stage;
- automatic dependency inference without `depends_on`.
Narratio should continue to orchestrate a fixed pipeline. The configurable part is the set of Scriptorium artifact invocations performed during the `analyze` stage.
## Current State
Narratio already has several relevant pieces in place:
- `pipeline.scriptorium.artifacts` is modeled as a map of artifact definitions.
- The Scriptorium adapter already accepts generic run/render requests.
- The artifact resolver already understands canonical artifact source IDs.
- The `analyze` stage already resolves inputs, optionally runs render-debug, invokes Scriptorium, verifies output, and records metadata.
The main limitation is that `analyze` currently treats `session_recap` as the only executable artifact and rejects other enabled artifact definitions.
## Target Architecture
### Runtime Artifact Catalog
Introduce a per-run artifact catalog that tracks built-in artifacts and configured artifacts.
Conceptually:
```text
ArtifactCatalog
├── built-in artifacts
│ ├── narratio.transcript.merged
│ ├── narratio.transcript.polished
│ ├── narratio.transcript.full
│ ├── narratio.transcript.trimmed
│ └── narratio.bounds.session
└── configured artifacts
├── narratio.artifact.session_recap
├── narratio.artifact.player_handout
└── narratio.artifact.npc_summary
```
The catalog should distinguish between three states:
```text
planned valid configured or built-in artifact known to Narratio
available artifact has been produced or otherwise resolved
executable configured artifact selected for execution in this analyze run
```
Configured artifacts can be planned without being executable. This distinction is important for disabled artifacts and for `--artifacts` filtering.
### Configured Artifact Source IDs
Configured artifact keys map directly to source IDs:
```text
pipeline.scriptorium.artifacts.<name>
→ narratio.artifact.<name>
```
`session_recap` should no longer be a special built-in analyze artifact. Instead, it is just a conventional configured artifact key:
```yaml
scriptorium:
artifacts:
session_recap:
enabled: true
prompt_id: dnd_session.session_recap
output_path: artifacts/session_recap.md
```
`narratio.artifact.session_recap` remains valid only because `session_recap` is configured.
### Artifact Output Directory
Add an internal default artifact output directory, initially:
```text
artifacts
```
This default should live in `internal/config/defaults.go` or the existing equivalent defaults location.
For the initial implementation:
- expose no public config knob for the artifact output directory;
- require each configured artifact to provide an explicit `output_path`;
- validate that each configured artifact `output_path` is run-relative;
- validate that each configured artifact `output_path` is under the internal artifact output directory;
- reject output paths that escape the run workspace or use path traversal.
This preserves future configurability without forcing Narratio to guess output extensions or formats now.
### Enabled, Disabled, and Selected Artifacts
Configured artifacts should have three distinct execution states:
```text
enabled by config artifact has enabled: true
selected for execution artifact remains executable after --artifacts filtering
disabled for execution artifact is not executable, but may be resolvable from disk
```
Without `--artifacts`, all configured artifacts with `enabled: true` are selected for execution.
With `--artifacts`, only the named artifacts are selected for execution. All other configured artifacts are treated as disabled for the current analyze invocation, regardless of their configured `enabled` value.
Disabled artifacts may still be resolved as inputs if their configured `output_path` exists on disk and passes validation.
### Disabled Artifact Resolution
If artifact `B` references artifact `A`, and `A` is disabled for execution, Narratio should attempt to resolve `A` from disk.
This should succeed only when:
1. `A` is defined in `pipeline.scriptorium.artifacts`;
2. `A` has a valid `output_path`;
3. the output path exists in the current run workspace;
4. the output is non-empty, or otherwise passes any available artifact-specific validation.
The resolved provenance should make the source clear, for example:
```text
filesystem.disabled_artifact_output
```
If the file does not exist or fails validation, the dependent artifact should fail before invoking Scriptorium.
Example error wording:
```text
artifact player_handout requires narratio.artifact.session_recap, but session_recap is disabled for execution and artifacts/session_recap.md does not exist
```
### Explicit Dependencies
Artifact-to-artifact references require explicit `depends_on` entries.
If artifact `B` has an input source of `narratio.artifact.A`, then `B.depends_on` must include `A`.
This should fail:
```yaml
scriptorium:
artifacts:
player_handout:
enabled: true
prompt_id: dnd_session.player_handout
output_path: artifacts/player_handout.md
inputs:
recap:
source: narratio.artifact.session_recap
required: true
```
This should pass:
```yaml
scriptorium:
artifacts:
player_handout:
enabled: true
depends_on:
- session_recap
prompt_id: dnd_session.player_handout
output_path: artifacts/player_handout.md
inputs:
recap:
source: narratio.artifact.session_recap
required: true
```
`depends_on` values refer to configured artifact keys, not full source IDs.
Dependency validation should fail on:
- references to unknown artifact keys;
- missing `depends_on` entries for artifact-to-artifact input references;
- self-dependencies;
- dependency cycles among executable artifacts.
Dependencies on disabled artifacts are permitted, but the disabled dependency must resolve from disk before the dependent artifact runs.
### Execution Order
The analyze stage should execute selected artifacts in dependency order.
Rules:
- selected artifacts are executable;
- disabled artifacts are never executed;
- selected artifacts may depend on other selected artifacts;
- selected artifacts may depend on disabled artifacts if those disabled artifacts resolve from disk;
- independent selected artifacts run in deterministic sorted-name order.
Use topological sorting over selected artifacts, while validating dependency references across the full configured artifact set.
### Input Resolution
Input resolution should use the artifact catalog and existing artifact resolver behavior.
For each configured artifact input:
- built-in sources resolve through existing resolver behavior;
- `previous_session_artifact` preserves existing behavior;
- `narratio.artifact.<name>` resolves through the runtime artifact catalog;
- selected dependencies resolve after being produced earlier in the same analyze execution;
- disabled dependencies resolve from their configured output path on disk;
- optional missing inputs are omitted;
- required missing inputs fail before Scriptorium is invoked.
### Analyze Stage Generalization
The `analyze` stage should become the generic Scriptorium artifact stage.
High-level flow:
1. Load configured Scriptorium artifacts.
2. Apply the `--artifacts` filter, if present.
3. If no artifacts are selected for execution, return success metadata with `skipped=true`.
4. Build the runtime artifact catalog.
5. Validate artifact names, output paths, source IDs, dependencies, selected artifacts, and required fields.
6. Resolve any disabled dependencies that are required by selected artifacts.
7. Sort selected artifacts by dependency order.
8. For each selected artifact:
- resolve configured inputs;
- build the Scriptorium run request;
- optionally run Scriptorium render-debug;
- run Scriptorium;
- fail on validation-failed result;
- verify the output exists and is non-empty;
- record artifact output metadata;
- register `narratio.artifact.<name>` as available in the catalog.
9. Return aggregate analyze-stage metadata containing all generated and reused artifacts relevant to the run.
The Scriptorium adapter should remain generic. It should not decide which artifacts run, how dependencies work, or how artifacts are registered.
### Manifest and Metadata
The manifest should remain stage-oriented.
This means:
- `analyze` succeeds or fails as a full stage;
- if `analyze` has already succeeded and the user does not force it, the runner skips it as a full stage;
- Narratio should not implement per-artifact resume in the first version.
However, analyze-stage metadata should still record artifact outputs for provenance and future resolution.
Recommended metadata shape:
```json
{
"skipped": false,
"artifacts": [
{
"name": "session_recap",
"source_id": "narratio.artifact.session_recap",
"output_kind": "scriptorium_artifact",
"path": "artifacts/session_recap.md",
"prompt_id": "dnd_session.session_recap",
"profile_id": "local-gemma-31b",
"provenance": "generated.current_analyze_run"
},
{
"name": "player_handout",
"source_id": "narratio.artifact.player_handout",
"output_kind": "scriptorium_artifact",
"path": "artifacts/player_handout.md",
"prompt_id": "dnd_session.player_handout",
"profile_id": "local-gemma-31b",
"provenance": "generated.current_analyze_run"
}
],
"reused_artifacts": [
{
"name": "session_recap",
"source_id": "narratio.artifact.session_recap",
"path": "artifacts/session_recap.md",
"provenance": "filesystem.disabled_artifact_output"
}
]
}
```
The exact struct can differ from this example, but it should preserve:
- artifact name;
- canonical source ID;
- output path;
- prompt/profile provenance for generated artifacts;
- reused-vs-generated provenance.
### Resume and Force Behavior
Keep resume behavior stage-level.
Recommended semantics:
```text
No --force, analyze already succeeded:
runner skips analyze, regardless of --artifacts.
--force, no --artifacts:
analyze regenerates all configured artifacts with enabled: true.
--force --artifacts player_handout:
analyze treats only player_handout as executable.
all other configured artifacts are disabled for execution.
disabled dependencies may be reused from disk.
--artifacts player_handout on a not-yet-completed analyze stage:
analyze runs only player_handout.
disabled dependencies may be reused from disk.
```
`--artifacts` should not imply `--force`. It is an execution filter, not a resume override.
### `--artifacts` CLI Flag
Add an `--artifacts` flag to commands that can execute or resume the analyze stage.
The flag should accept one or more configured artifact names. Internally, normalize values to a set of artifact keys.
Recommended behavior:
- validate all requested artifact names against `pipeline.scriptorium.artifacts`;
- reject unknown artifact names before running stages;
- treat requested artifacts as the only executable artifacts for the analyze stage;
- treat all other configured artifacts as disabled for execution;
- allow disabled artifacts to satisfy dependencies from disk as described above;
- if `--artifacts` is used while executing a stage other than `analyze`, either reject it or ignore it with a clear validation error. Prefer rejection.
The exact CLI parsing style can follow Narratio's existing conventions. Both comma-separated and repeatable values are acceptable if the CLI package supports them cleanly, but the internal representation should be a set of artifact keys.
### Archive Behavior
Do not automatically archive every generated artifact.
Artifact generation and archive promotion should remain separate concerns. Operators should continue to use `archive.promote_artifacts` to decide which generated files should be promoted or uploaded.
Example:
```yaml
archive:
promote_artifacts:
- from: artifacts/session_recap.md
to: artifacts/session_recap.md
required: true
- from: artifacts/player_handout.md
to: artifacts/player_handout.md
required: false
```
A later enhancement may add opt-in automatic promotion of configured artifacts, but explicit promotion should remain the default.
## Implementation Plan
### Phase 1: Config Model and Defaults
Add or update the configured artifact model to include:
- `enabled`;
- `depends_on`;
- `prompt_id`;
- `profile_id`;
- `output_path`;
- `timeout`;
- `render_debug`;
- `inputs`;
- `vars`.
Add an internal default artifact output directory in `internal/config/defaults.go`, initially set to `artifacts`.
Validation rules:
- artifact names must match a conservative identifier pattern such as `^[a-z][a-z0-9_]*$`;
- selected/executable artifacts require `prompt_id` and `output_path`;
- configured artifacts that may be referenced while disabled require `output_path`;
- configured artifact output paths must be run-relative;
- configured artifact output paths must live under the internal artifact output directory;
- configured artifact output paths must not escape the run workspace;
- `narratio.artifact.<name>` input sources must refer to configured artifact keys;
- any `narratio.artifact.<name>` input source must have a matching `depends_on` entry;
- `depends_on` entries must refer to configured artifact keys;
- dependencies must not contain self-references or executable cycles;
- input names and var names must remain compatible with the Scriptorium adapter's validation rules;
- unknown YAML fields must continue to fail strict decode.
Tests:
- valid single configured artifact;
- valid multiple independent artifacts;
- valid artifact-to-artifact dependency;
- valid dependency on disabled artifact with output path;
- invalid artifact name;
- missing required fields;
- output path outside `artifacts/`;
- dependency on missing artifact;
- missing `depends_on` for artifact input source;
- self-dependency;
- cycle detection;
- typo in `narratio.artifact.<name>` source;
- unknown YAML fields still fail strict decode.
### Phase 2: CLI Filtering
Add the `--artifacts` flag and carry the selected artifact set into the run execution options.
Implementation notes:
- parse values according to existing CLI conventions;
- normalize to artifact key strings;
- validate against configured artifact definitions after config load;
- make the selected set available to the analyze stage;
- reject use with commands or stages where analyze cannot run.
Tests:
- no `--artifacts` means all enabled artifacts are selected;
- one requested artifact is selected;
- multiple requested artifacts are selected;
- unknown requested artifact fails;
- `--artifacts` does not imply `--force`;
- `--artifacts` with already-succeeded analyze stage is skipped unless forced;
- `--artifacts` on unsupported stage command fails clearly.
### Phase 3: Runtime Artifact Catalog
Introduce an internal artifact catalog abstraction.
Responsibilities:
- register built-in artifact definitions;
- register configured artifact definitions;
- map configured artifact keys to `narratio.artifact.<name>` IDs;
- track planned, available, and executable artifact states;
- expose lookup by canonical source ID;
- record generated provenance;
- record disabled-from-disk provenance.
Keep the catalog narrow. It should not execute Scriptorium and should not understand prompt semantics.
Tests:
- built-in source lookup;
- configured source registration;
- duplicate/conflicting source handling;
- planned but unavailable artifact lookup;
- selected artifact state;
- disabled artifact state;
- registering an artifact as available after generation;
- registering a disabled artifact as available from disk;
- resolving a configured artifact from analyze metadata if that behavior is implemented.
### Phase 4: Resolver Integration
Update artifact resolution so configured artifact IDs are resolved through the runtime catalog.
Resolution behavior:
- built-in sources continue using existing resolver behavior;
- configured artifact sources resolve from catalog availability/provenance;
- selected configured artifacts become available after generation;
- disabled configured artifacts may become available from disk;
- missing optional configured artifact inputs are omitted;
- missing required configured artifact inputs fail clearly.
Tests:
- configured artifact consumes a built-in transcript source;
- configured artifact consumes another configured artifact produced earlier in the same analyze run;
- configured artifact consumes a disabled artifact resolved from disk;
- required disabled artifact missing on disk fails;
- required configured artifact missing fails;
- optional missing configured artifact is omitted;
- reused artifact provenance is recorded distinctly from generated artifact provenance.
### Phase 5: Analyze Stage Generalization
Refactor `analyze` to execute selected configured artifacts.
Implementation notes:
- remove the hard-coded `session_recap` selection path;
- remove the hard-coded rejection of non-`session_recap` artifacts;
- preserve skip behavior when Scriptorium config is absent or no artifacts are selected;
- build the runtime artifact catalog;
- apply `--artifacts` filtering;
- validate selected artifacts and their dependencies;
- pre-resolve disabled dependencies from disk where required;
- compute deterministic dependency order;
- execute selected artifacts one at a time in dependency order;
- keep render-debug behavior at global and artifact levels;
- keep Scriptorium adapter invocation generic;
- after each successful run, register the artifact as available in the catalog;
- aggregate generated and reused artifact metadata.
Tests:
- no Scriptorium config skips;
- empty artifact map skips;
- no selected artifacts skips;
- disabled artifacts do not run;
- one selected artifact runs;
- multiple independent artifacts run in deterministic order;
- dependent selected artifact receives prior selected artifact as input;
- dependent selected artifact receives disabled-from-disk artifact as input;
- render-debug works for configured artifacts;
- Scriptorium validation failure fails the stage;
- missing required input fails the stage;
- successful outputs are non-empty and recorded;
- artifact filter executes only requested artifacts.
### Phase 6: Manifest and Stage Metadata
Update analyze-stage metadata and manifest output recording to support dynamic configured artifacts.
Recommended behavior:
- every generated configured artifact gets `source_id: narratio.artifact.<name>`;
- every generated configured artifact gets a generic output kind such as `scriptorium_artifact`;
- reused disabled artifacts are recorded separately from generated artifacts;
- metadata is sufficient for debugging, provenance, and future resolver support;
- metadata does not create per-artifact resume semantics.
Because this is a pre-release hard cutover, do not preserve a special legacy `session_recap` output kind unless a current internal test or archive path still requires it temporarily. Prefer updating tests and examples to treat `session_recap` as an ordinary configured artifact.
Tests:
- metadata records one generated configured artifact;
- metadata records multiple generated configured artifacts;
- metadata records reused disabled artifact provenance;
- `session_recap` is recorded as a normal configured artifact;
- manifest still treats `analyze` as a single succeeded or failed stage;
- runner skip behavior remains stage-level.
### Phase 7: Archive and Promotion Review
Review archive behavior after dynamic artifacts are recorded.
Implementation notes:
- do not automatically promote every configured artifact;
- keep `archive.promote_artifacts` explicit;
- update default or example promotion rules to use configured `session_recap` output path;
- ensure required promotion rules fail clearly when selected artifact generation did not produce a required file.
Tests:
- generated artifact can be promoted by explicit archive rule;
- required archive promotion fails if selected artifact was not generated and no file exists;
- optional archive promotion skips cleanly if file is absent;
- hard cutover does not rely on hard-coded `session_recap` generation.
### Phase 8: Documentation and Examples
Status: complete.
Update documentation after the implementation is complete.
Recommended documentation changes:
- update `docs/config.md` with the generalized artifact configuration model;
- update `docs/internal/artifacts.md` to describe the runtime artifact catalog;
- update `docs/stages/analyze.md` to describe generic Scriptorium artifact generation;
- update Scriptorium integration docs only if the adapter contract changes;
- update full annotated pipeline examples;
- add at least one example with multiple artifacts and one dependency;
- document `--artifacts` behavior and its relationship to `--force`;
- remove documentation stating that only `session_recap` is supported.
Documentation should make clear that:
- configured artifact source IDs use `narratio.artifact.<name>`;
- `depends_on` uses artifact keys, not full source IDs;
- artifact-to-artifact source references require explicit `depends_on`;
- disabled artifacts can be reused from disk when required by selected artifacts;
- `--artifacts` filters execution but does not imply `--force`;
- archive promotion remains explicit;
- per-artifact resume is not part of the initial implementation.
## Migration Strategy
Because Narratio is pre-release, perform a hard cutover.
Required changes:
1. Remove the hard-coded `session_recap` analyze behavior.
2. Require `session_recap` to be declared under `pipeline.scriptorium.artifacts.session_recap` if the operator wants a session recap.
3. Treat `narratio.artifact.session_recap` as valid only when `session_recap` is a configured artifact key.
4. Update config examples to show `session_recap` as a normal configured artifact.
5. Update tests to stop assuming that `session_recap` is a built-in analyze artifact.
6. Keep archive promotion explicit and path-based.
Example replacement config:
```yaml
scriptorium:
binary: scriptorium
config_path: /etc/scriptorium/config.yml
timeout: 10m
render_debug: false
artifacts:
session_recap:
enabled: true
prompt_id: dnd_session.session_recap
profile_id: local-gemma-31b
output_path: artifacts/session_recap.md
timeout: 20m
inputs:
transcript:
source: narratio.transcript.trimmed
required: true
prior_recap:
source: previous_session_artifact
artifact: artifacts/session_recap.md
required: false
vars:
artifact_title: Session Recap
```
## Acceptance Criteria
The feature is complete when:
- operators can define more than one enabled Scriptorium artifact in `pipeline.yml`;
- Narratio runs selected artifacts in deterministic dependency order;
- configured artifacts are addressable as `narratio.artifact.<name>`;
- one configured artifact can consume another configured artifact as an input;
- artifact-to-artifact input references require explicit `depends_on`;
- disabled artifacts can satisfy dependencies from existing on-disk outputs;
- missing required disabled artifacts fail clearly;
- optional missing inputs are omitted;
- `--artifacts` can selectively execute valid configured artifact names;
- `--artifacts` does not imply `--force`;
- render-debug behavior works for all configured artifacts;
- generated and reused artifacts are recorded in analyze-stage metadata;
- `session_recap` is no longer hard-coded and works as a normal configured artifact;
- archive promotion remains explicit;
- tests cover config validation, dependency sorting, disabled artifact resolution, resolver behavior, CLI filtering, analyze execution, archive interactions, and metadata.
## Suggested Implementation Order
1. Config model, defaults, and validation.
2. CLI parsing and propagation of `--artifacts` selection.
3. Runtime artifact catalog.
4. Resolver integration for configured artifacts.
5. Analyze stage generalization.
6. Stage metadata and manifest output recording.
7. Archive behavior review.
8. Documentation and examples.
This order keeps the most static pieces first, then moves into execution behavior once the configuration contract is explicit and well tested.

289
docs/troubleshooting.md Normal file
View File

@@ -0,0 +1,289 @@
# Troubleshooting
## Purpose
Canonical operator troubleshooting guide for recurring implemented Narratio failures.
## Config file discovery failure
Symptom:
- `run`, `plan`, `resume`, or `run-stage` fails with config/session not found.
Likely Cause:
- `pipeline.yml` or `session.yml` is missing from discovery paths.
- wrong working directory when relying on `./session.yml`.
Diagnostics:
```bash
pwd
ls -l ./session.yml
ls -l /usr/local/etc/narratio/pipeline.yml /etc/narratio/pipeline.yml
```
Safe Fix:
- pass explicit `--config` and `--session`.
- or place files in documented discovery paths.
Links:
- [docs/config.md](./config.md)
- [docs/cli.md](./cli.md)
## Session template rendering failure
Symptom:
- load fails with unresolved placeholder or `session_id` mismatch.
Likely Cause:
- templated `session.yml` used without `--session-id`.
- rendered `session_id` differs from passed `--session-id`.
Diagnostics:
```bash
narratio plan --session ./session.yml --session-id 2026-04-04
```
Safe Fix:
- pass `--session-id` when template placeholders are present.
- ensure rendered `session_id` matches intended run session id.
Links:
- [docs/config.md](./config.md)
## Strict YAML decode or validation failure
Symptom:
- config load fails with unknown field or validation error.
Likely Cause:
- typo/stale field name.
- missing required fields or invalid constraints.
Diagnostics:
```bash
narratio plan --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04
```
Safe Fix:
- align fields/values to canonical config reference and examples.
Links:
- [docs/config.md](./config.md)
- [examples/](../examples/)
## `--artifacts` selection failure
Symptom:
- `run`/`resume`/`run-stage` fails with invalid or unknown artifact selection.
Likely Cause:
- `--artifacts` contains blank names or unknown artifact keys.
- `pipeline.scriptorium.artifacts` missing while using `--artifacts`.
Diagnostics:
```bash
narratio run --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --artifacts player_handout
```
Safe Fix:
- use configured artifact keys only.
- ensure `pipeline.scriptorium.artifacts` is defined.
Links:
- [docs/cli.md](./cli.md)
- [docs/config.md](./config.md)
## `run-stage --artifacts` on non-analyze stage
Symptom:
- `run-stage` fails with `--artifacts is only supported for stage "analyze"`.
Likely Cause:
- `--artifacts` was used with a non-`analyze` stage.
Diagnostics:
```bash
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --artifacts session_recap polish
```
Safe Fix:
- use `--artifacts` only with `run-stage ... analyze`.
Links:
- [docs/cli.md](./cli.md)
## Configured artifact dependency/input validation failure
Symptom:
- config validation fails for `depends_on`, `narratio.artifact.<name>` source, or artifact output path.
Likely Cause:
- `narratio.artifact.<name>` source missing matching `depends_on` key.
- dependency references unknown artifact key.
- dependency self-reference or enabled dependency cycle.
- artifact output path missing/invalid/outside `artifacts/` root.
Diagnostics:
```bash
narratio plan --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04
```
Safe Fix:
- ensure artifact-to-artifact inputs have explicit `depends_on` entries using artifact keys.
- ensure referenced artifacts exist and define valid `output_path` values.
- keep output paths relative and under `artifacts/`.
Links:
- [docs/config.md](./config.md)
- [docs/internal/stage-analyze.md](./internal/stage-analyze.md)
## Required configured artifact input unavailable at analyze time
Symptom:
- analyze fails because configured input source is unavailable.
Likely Cause:
- required upstream configured artifact was not selected/executed this run.
- non-executable dependency output file is missing or invalid on disk.
Diagnostics:
```bash
narratio status --manifest /path/to/manifest.json
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --artifacts player_handout analyze
```
Safe Fix:
- run analyze with needed artifacts selected.
- or ensure dependency output file exists at configured path and is valid.
Links:
- [docs/operations.md](./operations.md)
- [docs/config.md](./config.md)
## Manifest/status path failure
Symptom:
- `status` fails because manifest path is missing, unreadable, or invalid.
Likely Cause:
- wrong manifest path.
- manifest removed after cleanup.
- `--manifest` omitted.
Diagnostics:
```bash
narratio status --manifest /path/to/manifest.json
ls -l /path/to/manifest.json
```
Safe Fix:
- use manifest path printed by `run`, `resume`, or `run-stage`.
Links:
- [docs/cli.md](./cli.md)
- [docs/operations.md](./operations.md)
## Session lock conflict (`.lock`)
Symptom:
- run fails with lock conflict for session workdir.
Likely Cause:
- another Narratio process is running same session.
- stale lock from interrupted prior run.
Diagnostics:
```bash
ls -l {workspace.root}/work/{campaign}/{session_id}/.lock
cat {workspace.root}/work/{campaign}/{session_id}/.lock
ps aux | grep narratio
```
Safe Fix:
- wait for active run to finish.
- if no process is active, remove only stale session `.lock` file.
Links:
- [docs/operations.md](./operations.md)
- [docs/internal/workspace.md](./internal/workspace.md)
## Secrets env-dir or credential-env failure
Symptom:
- startup fails loading secrets directory, or stage fails due to missing credential env vars.
Likely Cause:
- invalid `pipeline.secrets.env_dir` path/permissions.
- required credential env var unset/empty.
Diagnostics:
```bash
ls -la /path/to/secrets_dir
env | grep -E 'AUDITA|OBJECT_STORAGE|AWS|SCRIPTORIUM'
```
Safe Fix:
- fix secrets directory and credential env vars.
- keep secret values out of YAML.
Links:
- [docs/config.md](./config.md)
## S3-audio prepare failure
Symptom:
- `prepare` fails in S3 mode (listing/downloading/no audio/backend error).
Likely Cause:
- wrong `session.inputs.audio_s3.prefix`.
- no `.flac` files at resolved prefix.
- invalid/missing object-store credentials or backend config.
- mixed local+S3 audio input config.
Diagnostics:
```bash
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 prepare
```
Safe Fix:
- configure exactly one audio source mode.
- verify `.flac` files and storage access.
Links:
- [docs/config.md](./config.md)
- [docs/operations.md](./operations.md)
## Archive promotion/current-pointer failure
Symptom:
- archive fails on required promotion source missing or pointer write failure.
Likely Cause:
- required promoted file absent (including analyze outputs not generated for this run).
- storage upload failed before `current/run_id.txt` commit marker write.
Diagnostics:
```bash
narratio status --manifest /path/to/manifest.json
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 archive
```
Safe Fix:
- rerun or resume upstream stages to generate required files.
- adjust promotion rules to match files that must exist.
- retry after storage issue is resolved.
Links:
- [docs/operations.md](./operations.md)
- [docs/config.md](./config.md)
- [docs/internal/stage-archive.md](./internal/stage-archive.md)

View File

@@ -0,0 +1,182 @@
# Full annotated pipeline example for implemented Narratio config fields.
# Values are safe placeholders and must be adapted per environment.
workspace:
# Optional: defaults to /var/lib/narratio.
root: /var/lib/narratio/workspace
# Optional: remove run-scoped workdir after successful archive commit.
cleanup_after_archive: false
# Optional: local secret file loader (directory of ENV_VAR_NAME files).
# secrets:
# env_dir: ./secrets
storage:
# Optional storage backend selector; use "s3" for archive + S3 audio workflows.
backend: s3
# Compatibility fields retained in schema.
bucket: ""
prefix: ""
s3:
# Required when using S3 audio or S3 archive uploads.
bucket: my-dnd-archive
# Optional; defaults to "dnd".
root_prefix: dnd
# Optional region/endpoint settings.
region: us-east-1
endpoint: ""
force_path_style: false
# Optional; defaults shown explicitly.
access_key_id_env: OBJECT_STORAGE_KEY_ID
secret_access_key_env: OBJECT_STORAGE_KEY
spool:
# Optional; defaults to /var/spool/narratio.
root: /var/spool/narratio
# Optional cleanup of run-scoped spool audio after successful archive commit.
delete_audio_after_archive: false
archive:
# Optional booleans; defaults are true.
enabled: true
upload_run: true
# Optional promotion rules; required files fail archive if missing.
promote_artifacts:
- from: transcripts/trimmed.json
to: transcripts/trimmed.json
required: true
- from: artifacts/session_recap.md
to: artifacts/session_recap.md
required: true
- from: artifacts/player_handout.md
to: artifacts/player_handout.md
required: false
whisperx:
# Required.
transcribe_url: "https://transcription.example.com/transcribe"
# Optional overrides; defaults shown explicitly.
language: en
timeout: 30m
retries: 3
retry_delay: 2s
concurrency: 2
seriatim:
# Optional overrides; defaults shown explicitly.
binary: seriatim
timeout: 10m
output_schema: seriatim-intermediate
coalesce_gap: 3.0
report: true
env:
# Optional advanced tuning; set only when needed.
overlap_word_run_gap: 1.0
overlap_word_run_reorder_window: 1.0
backchannel_max_duration: 2.0
filler_max_duration: 1.25
audita:
# Optional overrides; defaults shown explicitly where applicable.
binary: audita
timeout: 3h
llm_api_key_env: AUDITA_LLM_API_KEY
modules: [glossary, homophones, spoken_word, grammar]
base_url: ""
model: ""
total_llm_concurrency: 2
proposal_llm_concurrency: 1
validation_model: ""
validation_llm_concurrency: 1
transcript_description: ""
config_path: /usr/local/etc/audita/config.yml
output_schema: audita-v1
work_dir_retention: auto
report: true
normalize:
# Optional; defaults shown explicitly.
output_path: transcripts/normalized.json
output_schema: seriatim-intermediate
report: true
trim:
# Keep disabled unless bounds prompt integration is configured.
enabled: false
output_path: transcripts/trimmed.json
bounds:
prompt_id: dnd.session_bounds
profile_id: local-fast
transcript_input_name: transcript
output_path: reports/session_bounds.json
timeout: 10m
render_debug: false
render_output_path: reports/session_bounds.render.json
seriatim:
report: false
scriptorium:
binary: scriptorium
config_path: /usr/local/etc/scriptorium/config.yml
timeout: 10m
render_debug: false
artifacts:
# Configured artifact keys map to source IDs narratio.artifact.<key>.
session_recap:
enabled: true
prompt_id: dnd.session_recap
profile_id: local-fast
output_path: artifacts/session_recap.md
timeout: 10m
inputs:
transcript:
source: narratio.transcript.trimmed
required: true
previous_recap:
source: previous_session_artifact
artifact: session_recap
path: ""
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: session_recap
# Example dependent artifact:
# - depends_on entries use artifact keys.
# - narratio.artifact.<key> sources require matching depends_on membership.
player_handout:
enabled: true
depends_on:
- session_recap
prompt_id: dnd.player_handout
profile_id: local-fast
output_path: artifacts/player_handout.md
timeout: 10m
inputs:
recap:
source: narratio.artifact.session_recap
required: true
transcript:
source: narratio.transcript.trimmed
required: true
vars:
session_id: true
campaign_name: true
output_kind: player_handout
analyzer:
# Optional adapter settings.
binary_path: ""
timeout: 2m
artifacts:
output_dir: ""
types: []
notification:
# Optional notification settings.
backend: ""
recipient: ""
timeout: 30s

View File

@@ -1,109 +1,2 @@
workspace:
root: ./tmp/narratio-workspace
storage:
backend: local
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"
language: "en"
timeout: "30m"
retries: 3
retry_delay: "2s"
concurrency: 2
seriatim:
binary: "seriatim"
timeout: "10m"
output_schema: "seriatim-intermediate"
coalesce_gap: 3.0
report: true
env:
overlap_word_run_gap: 1.0
overlap_word_run_reorder_window: 1.0
backchannel_max_duration: 2.0
filler_max_duration: 1.25
audita:
binary: "audita"
timeout: "3h"
llm_api_key_env: "AUDITA_LLM_API_KEY"
modules:
- glossary
- homophones
- glossary
- spoken_word
- grammar
- homophones
- glossary
base_url: "https://openrouter.ai/api/v1"
model: "openrouter/google/gemma-4-31b-it"
llm_concurrency: 1
validation_model: ""
validation_llm_concurrency: 1
report: true
normalize:
# Session-workdir-relative when not absolute.
output_path: "transcripts/normalized.json"
output_schema: "seriatim-intermediate"
report: true
trim:
enabled: true
# Session-workdir-relative when not absolute.
output_path: "transcripts/trimmed.json"
bounds:
prompt_id: "dnd_session.bounds"
# Empty means use prompt default profile.
profile_id: ""
transcript_input_name: "transcript"
output_path: "artifacts/session_bounds.json"
timeout: "10m"
render_debug: false
render_output_path: "artifacts/session_bounds.render.json"
seriatim:
report: false
scriptorium:
binary: "scriptorium"
config_path: "/etc/scriptorium/config.yml"
timeout: "10m"
render_debug: false
artifacts:
session_recap:
enabled: true
prompt_id: "dnd.session_recap"
profile_id: "local-quality"
output_path: "artifacts/session_recap.md"
timeout: "10m"
# Optional per-artifact override of global scriptorium.render_debug.
# render_debug: true
inputs:
transcript:
# Available transcript sources:
# - trimmed_transcript (recommended for session_recap)
# - normalized_transcript (recommended for future full-session analysis)
# - processed_transcript (raw Audita-polished output)
source: "trimmed_transcript"
required: true
previous_recap:
source: "previous_session_artifact"
artifact: "session_recap"
# Optional: set when previous recap is available.
path: ""
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: "session_recap"
analyzer:
timeout: 20m
artifacts:
output_dir: artifacts
notification:
timeout: 10s

View File

@@ -0,0 +1,116 @@
workspace:
root: /var/lib/narratio/workspace
cleanup_after_archive: true
storage:
backend: s3
s3:
bucket: my-dnd-archive
root_prefix: dnd
region: us-east-1
access_key_id_env: OBJECT_STORAGE_KEY_ID
secret_access_key_env: OBJECT_STORAGE_KEY
spool:
root: /var/spool/narratio
delete_audio_after_archive: true
archive:
enabled: true
upload_run: true
promote_artifacts:
- from: transcripts/trimmed.json
to: transcripts/trimmed.json
required: true
- from: artifacts/session_recap.md
to: artifacts/session_recap.md
required: true
- from: artifacts/player_handout.md
to: artifacts/player_handout.md
required: false
whisperx:
transcribe_url: "https://transcription.example.com/transcribe"
language: en
timeout: 45m
retries: 3
retry_delay: 3s
concurrency: 2
seriatim:
binary: seriatim
timeout: 10m
output_schema: seriatim-intermediate
coalesce_gap: 3.0
report: true
audita:
binary: audita
timeout: 3h
llm_api_key_env: AUDITA_LLM_API_KEY
modules: [glossary, homophones, spoken_word, grammar]
output_schema: audita-v1
work_dir_retention: auto
total_llm_concurrency: 2
proposal_llm_concurrency: 1
validation_llm_concurrency: 1
report: true
normalize:
output_path: transcripts/normalized.json
output_schema: seriatim-intermediate
report: true
trim:
enabled: false
scriptorium:
binary: scriptorium
config_path: /usr/local/etc/scriptorium/config.yml
timeout: 10m
render_debug: false
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
profile_id: local-fast
output_path: artifacts/session_recap.md
timeout: 10m
inputs:
transcript:
source: narratio.transcript.trimmed
required: true
previous_recap:
source: previous_session_artifact
artifact: session_recap
required: false
vars:
session_id: true
session_date: true
campaign_name: true
previous_session_id: true
output_kind: session_recap
player_handout:
enabled: true
depends_on:
- session_recap
prompt_id: dnd.player_handout
profile_id: local-fast
output_path: artifacts/player_handout.md
timeout: 10m
inputs:
recap:
source: narratio.artifact.session_recap
required: true
transcript:
source: narratio.transcript.trimmed
required: true
vars:
session_id: true
output_kind: player_handout
analyzer:
timeout: 2m
notification:
timeout: 30s

View File

@@ -0,0 +1,9 @@
session_id: 2026-05-03
campaign: sample-campaign
date: 2026-05-03
title: Sample Session
inputs:
audio_dir: ./audio
speakers_file: ./examples/speakers.yml
autocorrect_file: ./examples/autocorrect.yml
glossary_file: ./examples/glossary.yml

View File

@@ -1,10 +0,0 @@
session_id: 2026-05-03
campaign: sample-campaign
date: 2026-05-03
title: Sample Session
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml

View File

@@ -0,0 +1,10 @@
session_id: 2026-05-03
campaign: sample-campaign
date: 2026-05-03
title: Sample Session
inputs:
audio_s3:
prefix: audio/
speakers_file: ./examples/speakers.yml
autocorrect_file: ./examples/autocorrect.yml
glossary_file: ./examples/glossary.yml

View File

@@ -0,0 +1,7 @@
session_id: "{{ session_id }}"
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./examples/speakers.yml
autocorrect_file: ./examples/autocorrect.yml
glossary_file: ./examples/glossary.yml

25
go.mod
View File

@@ -2,4 +2,27 @@ module gitea.maximumdirect.net/eric/narratio
go 1.25.0
require gopkg.in/yaml.v3 v3.0.1
require (
github.com/aws/aws-sdk-go-v2/config v1.32.17
github.com/aws/aws-sdk-go-v2/credentials v1.19.16
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0
github.com/aws/smithy-go v1.25.1
gopkg.in/yaml.v3 v3.0.1
)
require (
github.com/aws/aws-sdk-go-v2 v1.41.7 // indirect
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10 // indirect
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.23 // indirect
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23 // indirect
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23 // indirect
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.24 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.9 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.15 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.23 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.23 // indirect
github.com/aws/aws-sdk-go-v2/service/signin v1.0.11 // indirect
github.com/aws/aws-sdk-go-v2/service/sso v1.30.17 // indirect
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.35.21 // indirect
github.com/aws/aws-sdk-go-v2/service/sts v1.42.1 // indirect
)

36
go.sum
View File

@@ -1,3 +1,39 @@
github.com/aws/aws-sdk-go-v2 v1.41.7 h1:DWpAJt66FmnnaRIOT/8ASTucrvuDPZASqhhLey6tLY8=
github.com/aws/aws-sdk-go-v2 v1.41.7/go.mod h1:4LAfZOPHNVNQEckOACQx60Y8pSRjIkNZQz1w92xpMJc=
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10 h1:gx1AwW1Iyk9Z9dD9F4akX5gnN3QZwUB20GGKH/I+Rho=
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10/go.mod h1:qqY157uZoqm5OXq/amuaBJyC9hgBCBQnsaWnPe905GY=
github.com/aws/aws-sdk-go-v2/config v1.32.17 h1:FpL4/758/diKwqbytU0prpuiu60fgXKUWCpDJtApclU=
github.com/aws/aws-sdk-go-v2/config v1.32.17/go.mod h1:OXqUMzgXytfoF9JaKkhrOYsyh72t9G+MJH8mMRaexOE=
github.com/aws/aws-sdk-go-v2/credentials v1.19.16 h1:r3RJBuU7X9ibt8RHbMjWE6y60QbKBiII6wSrXnapxSU=
github.com/aws/aws-sdk-go-v2/credentials v1.19.16/go.mod h1:6cx7zqDENJDbBIIWX6P8s0h6hqHC8Avbjh9Dseo27ug=
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.23 h1:UuSfcORqNSz/ey3VPRS8TcVH2Ikf0/sC+Hdj400QI6U=
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.23/go.mod h1:+G/OSGiOFnSOkYloKj/9M35s74LgVAdJBSD5lsFfqKg=
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23 h1:GpT/TrnBYuE5gan2cZbTtvP+JlHsutdmlV2YfEyNde0=
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23/go.mod h1:xYWD6BS9ywC5bS3sz9Xh04whO/hzK2plt2Zkyrp4JuA=
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23 h1:bpd8vxhlQi2r1hiueOw02f/duEPTMK59Q4QMAoTTtTo=
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23/go.mod h1:15DfR2nw+CRHIk0tqNyifu3G1YdAOy68RftkhMDDwYk=
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.24 h1:OQqn11BtaYv1WLUowvcA30MpzIu8Ti4pcLPIIyoKZrA=
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.24/go.mod h1:X5ZJyfwVrWA96GzPmUCWFQaEARPR7gCrpq2E92PJwAE=
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.9 h1:FLudkZLt5ci0ozzgkVo8BJGwvqNaZbTWb3UcucAateA=
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.9/go.mod h1:w7wZ/s9qK7c8g4al+UyoF1Sp/Z45UwMGcqIzLWVQHWk=
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.15 h1:ieLCO1JxUWuxTZ1cRd0GAaeX7O6cIxnwk7tc1LsQhC4=
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.15/go.mod h1:e3IzZvQ3kAWNykvE0Tr0RDZCMFInMvhku3qNpcIQXhM=
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.23 h1:pbrxO/kuIwgEsOPLkaHu0O+m4fNgLU8B3vxQ+72jTPw=
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.23/go.mod h1:/CMNUqoj46HpS3MNRDEDIwcgEnrtZlKRaHNaHxIFpNA=
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.23 h1:03xatSQO4+AM1lTAbnRg5OK528EUg744nW7F73U8DKw=
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.23/go.mod h1:M8l3mwgx5ToK7wot2sBBce/ojzgnPzZXUV445gTSyE8=
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0 h1:etqBTKY581iwLL/H/S2sVgk3C9lAsTJFeXWFDsDcWOU=
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0/go.mod h1:L2dcoOgS2VSgbPLvpak2NyUPsO1TBN7M45Z4H7DlRc4=
github.com/aws/aws-sdk-go-v2/service/signin v1.0.11 h1:TdJ+HdzOBhU8+iVAOGUTU63VXopcumCOF1paFulHWZc=
github.com/aws/aws-sdk-go-v2/service/signin v1.0.11/go.mod h1:R82ZRExE/nheo0N+T8zHPcLRTcH8MGsnR3BiVGX0TwI=
github.com/aws/aws-sdk-go-v2/service/sso v1.30.17 h1:7byT8HUWrgoRp6sXjxtZwgOKfhss5fW6SkLBtqzgRoE=
github.com/aws/aws-sdk-go-v2/service/sso v1.30.17/go.mod h1:xNWknVi4Ezm1vg1QsB/5EWpAJURq22uqd38U8qKvOJc=
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.35.21 h1:+1Kl1zx6bWi4X7cKi3VYh29h8BvsCoHQEQ6ST9X8w7w=
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.35.21/go.mod h1:4vIRDq+CJB2xFAXZ+YgGUTiEft7oAQlhIs71xcSeuVg=
github.com/aws/aws-sdk-go-v2/service/sts v1.42.1 h1:F/M5Y9I3nwr2IEpshZgh1GeHpOItExNM9L1euNuh/fk=
github.com/aws/aws-sdk-go-v2/service/sts v1.42.1/go.mod h1:mTNxImtovCOEEuD65mKW7DCsL+2gjEH+RPEAexAzAio=
github.com/aws/smithy-go v1.25.1 h1:J8ERsGSU7d+aCmdQur5Txg6bVoYelvQJgtZehD12GkI=
github.com/aws/smithy-go v1.25.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405 h1:yhCVgyC4o1eVCa2tZl7eS0r+SDo693bJlVdllGtEeKM=
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
gopkg.in/yaml.v3 v3.0.1 h1:fxVm/GzAzEWqLHuvctI91KS9hhNmmWOoWu0XTYJS7CA=

View File

@@ -24,6 +24,12 @@ type PolishRequest struct {
Modules []string
BaseURL string
Model string
TranscriptDescription string
ConfigPath string
OutputSchema string
WorkDirRetention string
TotalLLMConcurrency *int
ProposalLLMConcurrency *int
ValidationModel string
ValidationLLMConcurrency *int
StdoutLogPath string

View File

@@ -21,7 +21,12 @@ type SubprocessRunnerConfig struct {
Modules []string
BaseURL string
Model string
LLMConcurrency *int
TranscriptDescription string
ConfigPath string
OutputSchema string
WorkDirRetention string
TotalLLMConcurrency *int
ProposalLLMConcurrency *int
ValidationModel string
ValidationLLMConcurrency *int
Report bool
@@ -35,7 +40,12 @@ type SubprocessRunner struct {
modules []string
baseURL string
model string
llmConcurrency *int
transcriptDescription string
configPath string
outputSchema string
workDirRetention string
totalLLMConcurrency *int
proposalLLMConcurrency *int
validationModel string
validationLLMConcurrency *int
report bool
@@ -49,7 +59,12 @@ func NewSubprocessRunnerFromConfigValues(
modules []string,
baseURL string,
model string,
llmConcurrency *int,
transcriptDescription string,
configPath string,
outputSchema string,
workDirRetention string,
totalLLMConcurrency *int,
proposalLLMConcurrency *int,
validationModel string,
validationLLMConcurrency *int,
report bool,
@@ -68,7 +83,12 @@ func NewSubprocessRunnerFromConfigValues(
Modules: modules,
BaseURL: baseURL,
Model: model,
LLMConcurrency: llmConcurrency,
TranscriptDescription: transcriptDescription,
ConfigPath: configPath,
OutputSchema: outputSchema,
WorkDirRetention: workDirRetention,
TotalLLMConcurrency: totalLLMConcurrency,
ProposalLLMConcurrency: proposalLLMConcurrency,
ValidationModel: validationModel,
ValidationLLMConcurrency: validationLLMConcurrency,
Report: report,
@@ -83,33 +103,39 @@ func NewSubprocessRunner(cfg SubprocessRunnerConfig) (*SubprocessRunner, error)
if cfg.Timeout <= 0 {
return nil, fmt.Errorf("audita timeout must be > 0")
}
if len(cfg.Modules) == 0 {
return nil, fmt.Errorf("audita modules must include at least one module")
}
for i, module := range cfg.Modules {
if strings.TrimSpace(module) == "" {
return nil, fmt.Errorf("audita module at index %d is empty", i)
}
}
if strings.TrimSpace(cfg.BaseURL) == "" {
return nil, fmt.Errorf("audita base url is required")
}
u, err := url.Parse(cfg.BaseURL)
if err != nil || u.Scheme == "" || u.Host == "" {
if err != nil {
return nil, fmt.Errorf("audita base url %q is invalid: %w", cfg.BaseURL, err)
if strings.TrimSpace(cfg.BaseURL) != "" {
u, err := url.Parse(cfg.BaseURL)
if err != nil || u.Scheme == "" || u.Host == "" {
if err != nil {
return nil, fmt.Errorf("audita base url %q is invalid: %w", cfg.BaseURL, err)
}
return nil, fmt.Errorf("audita base url %q is invalid", cfg.BaseURL)
}
return nil, fmt.Errorf("audita base url %q is invalid", cfg.BaseURL)
}
if strings.TrimSpace(cfg.Model) == "" {
return nil, fmt.Errorf("audita model is required")
if cfg.TotalLLMConcurrency != nil && *cfg.TotalLLMConcurrency <= 0 {
return nil, fmt.Errorf("audita total llm concurrency must be > 0 when provided")
}
if cfg.LLMConcurrency != nil && *cfg.LLMConcurrency <= 0 {
return nil, fmt.Errorf("audita llm concurrency must be > 0 when provided")
if cfg.ProposalLLMConcurrency != nil && *cfg.ProposalLLMConcurrency <= 0 {
return nil, fmt.Errorf("audita proposal llm concurrency must be > 0 when provided")
}
if cfg.ValidationLLMConcurrency != nil && *cfg.ValidationLLMConcurrency <= 0 {
return nil, fmt.Errorf("audita validation llm concurrency must be > 0 when provided")
}
switch strings.TrimSpace(cfg.OutputSchema) {
case "", "bare-segments", "audita-v1":
default:
return nil, fmt.Errorf("audita output schema must be one of: bare-segments, audita-v1")
}
switch strings.TrimSpace(cfg.WorkDirRetention) {
case "", "always", "auto", "never":
default:
return nil, fmt.Errorf("audita work dir retention must be one of: always, auto, never")
}
modules := make([]string, len(cfg.Modules))
for i, m := range cfg.Modules {
@@ -123,7 +149,12 @@ func NewSubprocessRunner(cfg SubprocessRunnerConfig) (*SubprocessRunner, error)
modules: modules,
baseURL: strings.TrimSpace(cfg.BaseURL),
model: strings.TrimSpace(cfg.Model),
llmConcurrency: cfg.LLMConcurrency,
transcriptDescription: strings.TrimSpace(cfg.TranscriptDescription),
configPath: strings.TrimSpace(cfg.ConfigPath),
outputSchema: strings.TrimSpace(cfg.OutputSchema),
workDirRetention: strings.TrimSpace(cfg.WorkDirRetention),
totalLLMConcurrency: cfg.TotalLLMConcurrency,
proposalLLMConcurrency: cfg.ProposalLLMConcurrency,
validationModel: strings.TrimSpace(cfg.ValidationModel),
validationLLMConcurrency: cfg.ValidationLLMConcurrency,
report: cfg.Report,
@@ -152,7 +183,7 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
}
reqModules := req.Modules
if len(reqModules) == 0 {
if reqModules == nil {
reqModules = append([]string(nil), r.modules...)
}
args := r.buildArgs(req, reqModules)
@@ -168,14 +199,9 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
env["AUDITA_LLM_API_KEY"] = credential
credentialPresent = true
}
primaryConcurrencyViaEnv := false
if r.llmConcurrency != nil {
env["AUDITA_LLM_CONCURRENCY"] = strconv.Itoa(*r.llmConcurrency)
primaryConcurrencyViaEnv = true
}
if req.GeneratedConfigPath != "" {
if err := r.writeInvocationConfig(req, args, reqModules, credentialPresent, primaryConcurrencyViaEnv); err != nil {
if err := r.writeInvocationConfig(req, args, reqModules, credentialPresent); err != nil {
return PolishResult{}, fmt.Errorf("write audita invocation config %q: %w", req.GeneratedConfigPath, err)
}
}
@@ -196,7 +222,7 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
req.StderrLogPath,
)
wrappedMessage = addSubprocessStreamHint(wrappedMessage, err)
return r.failureResult(req, reqModules, runRes, credentialPresent, primaryConcurrencyViaEnv), fmt.Errorf(
return r.failureResult(req, reqModules, runRes, credentialPresent), fmt.Errorf(
"%s: %w",
wrappedMessage,
err,
@@ -204,11 +230,11 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
}
if err := validateProcessedOutput(req.OutputProcessedPath); err != nil {
return r.failureResult(req, reqModules, runRes, credentialPresent, primaryConcurrencyViaEnv), fmt.Errorf("validate audita processed output %q: %w", req.OutputProcessedPath, err)
return r.failureResult(req, reqModules, runRes, credentialPresent), fmt.Errorf("validate audita processed output %q: %w", req.OutputProcessedPath, err)
}
if r.report {
if err := validateJSONFile(req.ReportPath); err != nil {
return r.failureResult(req, reqModules, runRes, credentialPresent, primaryConcurrencyViaEnv), fmt.Errorf("validate audita report output %q: %w", req.ReportPath, err)
return r.failureResult(req, reqModules, runRes, credentialPresent), fmt.Errorf("validate audita report output %q: %w", req.ReportPath, err)
}
}
@@ -223,21 +249,25 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
Duration: runRes.Duration,
InvokedBinary: r.binary,
Metadata: map[string]any{
"adapter": "audita_subprocess",
"modules": reqModules,
"base_url": r.baseURL,
"model": r.model,
"validation_model": r.validationModel,
"validation_llm_concurrency": r.validationLLMConcurrency,
"credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent,
"primary_llm_concurrency_via_env": primaryConcurrencyViaEnv,
"primary_llm_concurrency_env_name": "AUDITA_LLM_CONCURRENCY",
"adapter": "audita_subprocess",
"modules": reqModules,
"base_url": r.baseURL,
"model": r.model,
"transcript_description": r.transcriptDescription,
"config_path": r.configPath,
"output_schema": r.outputSchema,
"work_dir_retention": r.workDirRetention,
"validation_model": r.validationModel,
"total_llm_concurrency": r.totalLLMConcurrency,
"proposal_llm_concurrency": r.proposalLLMConcurrency,
"validation_llm_concurrency": r.validationLLMConcurrency,
"credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent,
},
}, nil
}
func (r *SubprocessRunner) failureResult(req PolishRequest, modules []string, runRes subprocess.RunResult, credentialPresent bool, primaryConcurrencyViaEnv bool) PolishResult {
func (r *SubprocessRunner) failureResult(req PolishRequest, modules []string, runRes subprocess.RunResult, credentialPresent bool) PolishResult {
return PolishResult{
ProcessedTranscriptPath: req.OutputProcessedPath,
ReportPath: req.ReportPath,
@@ -249,16 +279,20 @@ func (r *SubprocessRunner) failureResult(req PolishRequest, modules []string, ru
Duration: runRes.Duration,
InvokedBinary: r.binary,
Metadata: map[string]any{
"adapter": "audita_subprocess",
"modules": modules,
"base_url": r.baseURL,
"model": r.model,
"validation_model": r.validationModel,
"validation_llm_concurrency": r.validationLLMConcurrency,
"credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent,
"primary_llm_concurrency_via_env": primaryConcurrencyViaEnv,
"primary_llm_concurrency_env_name": "AUDITA_LLM_CONCURRENCY",
"adapter": "audita_subprocess",
"modules": modules,
"base_url": r.baseURL,
"model": r.model,
"transcript_description": r.transcriptDescription,
"config_path": r.configPath,
"output_schema": r.outputSchema,
"work_dir_retention": r.workDirRetention,
"validation_model": r.validationModel,
"total_llm_concurrency": r.totalLLMConcurrency,
"proposal_llm_concurrency": r.proposalLLMConcurrency,
"validation_llm_concurrency": r.validationLLMConcurrency,
"credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent,
},
}
}
@@ -269,14 +303,38 @@ func (r *SubprocessRunner) buildArgs(req PolishRequest, modules []string) []stri
req.MergedTranscriptPath,
"--glossary", req.GlossaryPath,
"--output", req.OutputProcessedPath,
"--modules", strings.Join(modules, ","),
"--base-url", r.baseURL,
"--model", r.model,
"--work-dir", req.WorkDir,
}
if r.baseURL != "" {
args = append(args, "--base-url", r.baseURL)
}
if r.model != "" {
args = append(args, "--model", r.model)
}
if len(modules) > 0 {
args = append(args, "--modules", strings.Join(modules, ","))
}
if r.report {
args = append(args, "--report-json", req.ReportPath)
}
if r.transcriptDescription != "" {
args = append(args, "--transcript-description", r.transcriptDescription)
}
if r.configPath != "" {
args = append(args, "--config", r.configPath)
}
if r.outputSchema != "" {
args = append(args, "--output-schema", r.outputSchema)
}
if r.workDirRetention != "" {
args = append(args, "--work-dir-retention", r.workDirRetention)
}
if r.totalLLMConcurrency != nil {
args = append(args, "--total-llm-concurrency", strconv.Itoa(*r.totalLLMConcurrency))
}
if r.proposalLLMConcurrency != nil {
args = append(args, "--proposal-llm-concurrency", strconv.Itoa(*r.proposalLLMConcurrency))
}
if r.validationModel != "" {
args = append(args, "--validation-model", r.validationModel)
}
@@ -286,29 +344,31 @@ func (r *SubprocessRunner) buildArgs(req PolishRequest, modules []string) []stri
return args
}
func (r *SubprocessRunner) writeInvocationConfig(req PolishRequest, args []string, modules []string, credentialPresent bool, primaryConcurrencyViaEnv bool) error {
func (r *SubprocessRunner) writeInvocationConfig(req PolishRequest, args []string, modules []string, credentialPresent bool) error {
payload := map[string]any{
"schema": "audita.generated.v1",
"binary": r.binary,
"args": args,
"timeout": r.timeout.String(),
"modules": modules,
"base_url": r.baseURL,
"model": r.model,
"validation_model": r.validationModel,
"validation_llm_concurrency": r.validationLLMConcurrency,
"report_enabled": r.report,
"merged_transcript_path": req.MergedTranscriptPath,
"glossary_path": req.GlossaryPath,
"output_path": req.OutputProcessedPath,
"report_path": req.ReportPath,
"work_dir": req.WorkDir,
"credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent,
"primary_llm_concurrency_via_env": primaryConcurrencyViaEnv,
}
if r.llmConcurrency != nil {
payload["llm_concurrency"] = *r.llmConcurrency
"schema": "audita.generated.v1",
"binary": r.binary,
"args": args,
"timeout": r.timeout.String(),
"modules": modules,
"base_url": r.baseURL,
"model": r.model,
"transcript_description": r.transcriptDescription,
"config_path": r.configPath,
"output_schema": r.outputSchema,
"work_dir_retention": r.workDirRetention,
"validation_model": r.validationModel,
"total_llm_concurrency": r.totalLLMConcurrency,
"proposal_llm_concurrency": r.proposalLLMConcurrency,
"validation_llm_concurrency": r.validationLLMConcurrency,
"report_enabled": r.report,
"merged_transcript_path": req.MergedTranscriptPath,
"glossary_path": req.GlossaryPath,
"output_path": req.OutputProcessedPath,
"report_path": req.ReportPath,
"work_dir": req.WorkDir,
"credential_env_var": r.llmAPIKeyEnv,
"credential_present": credentialPresent,
}
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
}

View File

@@ -25,7 +25,8 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath)
wrapper := writeAuditaHelperWrapper(t)
llmConcurrency := 1
totalLLMConcurrency := 3
proposalLLMConcurrency := 2
validationLLMConcurrency := 2
runner, err := NewSubprocessRunner(SubprocessRunnerConfig{
Binary: wrapper,
@@ -34,7 +35,12 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
Modules: []string{"glossary", "homophones", "glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
TranscriptDescription: "Campaign Session 42",
ConfigPath: "/etc/audita/config.yml",
OutputSchema: "audita-v1",
WorkDirRetention: "auto",
TotalLLMConcurrency: &totalLLMConcurrency,
ProposalLLMConcurrency: &proposalLLMConcurrency,
ValidationModel: "openrouter/google/gemma-4-31b-it",
ValidationLLMConcurrency: &validationLLMConcurrency,
Report: true,
@@ -93,11 +99,17 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
"process", req.MergedTranscriptPath,
"--glossary", req.GlossaryPath,
"--output", req.OutputProcessedPath,
"--modules", "glossary,homophones,glossary",
"--work-dir", req.WorkDir,
"--base-url", "https://openrouter.ai/api/v1",
"--model", "openrouter/google/gemma-4-31b-it",
"--work-dir", req.WorkDir,
"--modules", "glossary,homophones,glossary",
"--report-json", req.ReportPath,
"--transcript-description", "Campaign Session 42",
"--config", "/etc/audita/config.yml",
"--output-schema", "audita-v1",
"--work-dir-retention", "auto",
"--total-llm-concurrency", "3",
"--proposal-llm-concurrency", "2",
"--validation-model", "openrouter/google/gemma-4-31b-it",
"--validation-llm-concurrency", "2",
}
@@ -107,8 +119,8 @@ func TestSubprocessRunnerSuccessArgsEnvAndValidation(t *testing.T) {
if rec.Env["AUDITA_LLM_API_KEY"] != "super-secret" {
t.Fatalf("AUDITA_LLM_API_KEY = %q, want propagated secret", rec.Env["AUDITA_LLM_API_KEY"])
}
if rec.Env["AUDITA_LLM_CONCURRENCY"] != "1" {
t.Fatalf("AUDITA_LLM_CONCURRENCY = %q, want 1", rec.Env["AUDITA_LLM_CONCURRENCY"])
if rec.Env["AUDITA_LLM_CONCURRENCY"] != "" {
t.Fatalf("AUDITA_LLM_CONCURRENCY = %q, want empty/omitted", rec.Env["AUDITA_LLM_CONCURRENCY"])
}
cfgData, err := os.ReadFile(req.GeneratedConfigPath)
@@ -124,16 +136,14 @@ func TestSubprocessRunnerMissingConfiguredCredentialFails(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "MISSING_AUDITA_KEY",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "MISSING_AUDITA_KEY",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: false,
})
req := auditaReqForTest(t, false)
@@ -155,16 +165,14 @@ func TestSubprocessRunnerUnconfiguredCredentialEnvOmitsCredential(t *testing.T)
recordPath := filepath.Join(t.TempDir(), "record.json")
t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath)
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: false,
})
req := auditaReqForTest(t, false)
@@ -211,6 +219,65 @@ func TestSubprocessRunnerInheritsParentEnvironment(t *testing.T) {
}
}
func TestSubprocessRunnerOmitsModulesFlagWhenNotConfigured(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_AUDITA_HELPER", "1")
t.Setenv("AUDITA_HELPER_MODE", "success")
recordPath := filepath.Join(t.TempDir(), "record.json")
t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath)
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "",
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: false,
})
req := auditaReqForTest(t, false)
if _, err := runner.Run(context.Background(), req); err != nil {
t.Fatalf("Run() error = %v", err)
}
rec := readAuditaHelperRecord(t, recordPath)
for i := 0; i < len(rec.Args); i++ {
if rec.Args[i] == "--modules" {
t.Fatalf("args contained --modules unexpectedly: %#v", rec.Args)
}
}
}
func TestSubprocessRunnerOmitsBaseURLAndModelFlagsWhenNotConfigured(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
}
t.Setenv("GO_WANT_AUDITA_HELPER", "1")
t.Setenv("AUDITA_HELPER_MODE", "success")
recordPath := filepath.Join(t.TempDir(), "record.json")
t.Setenv("AUDITA_HELPER_RECORD_PATH", recordPath)
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "",
Report: false,
})
req := auditaReqForTest(t, false)
if _, err := runner.Run(context.Background(), req); err != nil {
t.Fatalf("Run() error = %v", err)
}
rec := readAuditaHelperRecord(t, recordPath)
for i := 0; i < len(rec.Args); i++ {
if rec.Args[i] == "--base-url" {
t.Fatalf("args contained --base-url unexpectedly: %#v", rec.Args)
}
if rec.Args[i] == "--model" {
t.Fatalf("args contained --model unexpectedly: %#v", rec.Args)
}
}
}
func TestSubprocessRunnerSubprocessFailure(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("helper wrapper script uses /bin/sh")
@@ -220,16 +287,14 @@ func TestSubprocessRunnerSubprocessFailure(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: true,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: true,
})
req := auditaReqForTest(t, true)
_, err := runner.Run(context.Background(), req)
@@ -256,16 +321,14 @@ func TestSubprocessRunnerSubprocessFailureAddsStderrDescriptorHint(t *testing.T)
t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: true,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: true,
})
req := auditaReqForTest(t, true)
_, err := runner.Run(context.Background(), req)
@@ -286,16 +349,14 @@ func TestSubprocessRunnerMissingOutputFails(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: false,
})
req := auditaReqForTest(t, false)
_, err := runner.Run(context.Background(), req)
@@ -316,16 +377,14 @@ func TestSubprocessRunnerInvalidOutputJSONFails(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: false,
})
req := auditaReqForTest(t, false)
_, err := runner.Run(context.Background(), req)
@@ -346,16 +405,14 @@ func TestSubprocessRunnerSegmentsMissingFails(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: false,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: false,
})
req := auditaReqForTest(t, false)
_, err := runner.Run(context.Background(), req)
@@ -376,16 +433,14 @@ func TestSubprocessRunnerInvalidReportJSONFails(t *testing.T) {
t.Setenv("OPENAI_KEY_SOURCE", "super-secret")
t.Setenv("AUDITA_HELPER_RECORD_PATH", filepath.Join(t.TempDir(), "record.json"))
llmConcurrency := 1
runner := mustAuditaRunner(t, SubprocessRunnerConfig{
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: &llmConcurrency,
Report: true,
Binary: writeAuditaHelperWrapper(t),
Timeout: mustParseAuditaDuration(t, "2s"),
LLMAPIKeyEnv: "OPENAI_KEY_SOURCE",
Modules: []string{"glossary"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
Report: true,
})
req := auditaReqForTest(t, true)
_, err := runner.Run(context.Background(), req)
@@ -398,11 +453,11 @@ func TestSubprocessRunnerInvalidReportJSONFails(t *testing.T) {
}
func TestSubprocessRunnerConstructorValidation(t *testing.T) {
_, err := NewSubprocessRunnerFromConfigValues("", "3h", "AUDITA_LLM_API_KEY", []string{"glossary"}, "https://openrouter.ai/api/v1", "openrouter/google/gemma-4-31b-it", nil, "", nil, true)
_, err := NewSubprocessRunnerFromConfigValues("", "3h", "AUDITA_LLM_API_KEY", []string{"glossary"}, "https://openrouter.ai/api/v1", "openrouter/google/gemma-4-31b-it", "", "", "", "", nil, nil, "", nil, true)
if err == nil {
t.Fatal("expected binary validation error")
}
_, err = NewSubprocessRunnerFromConfigValues("audita", "bad", "AUDITA_LLM_API_KEY", []string{"glossary"}, "https://openrouter.ai/api/v1", "openrouter/google/gemma-4-31b-it", nil, "", nil, true)
_, err = NewSubprocessRunnerFromConfigValues("audita", "bad", "AUDITA_LLM_API_KEY", []string{"glossary"}, "https://openrouter.ai/api/v1", "openrouter/google/gemma-4-31b-it", "", "", "", "", nil, nil, "", nil, true)
if err == nil {
t.Fatal("expected timeout parse error")
}

View File

@@ -0,0 +1,29 @@
package storage
import (
"context"
"fmt"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
// NewObjectStoreFromConfig constructs a remote object store from resolved config.
func NewObjectStoreFromConfig(ctx context.Context, cfg *config.Config) (ObjectStore, error) {
if cfg == nil || cfg.Pipeline == nil {
return nil, fmt.Errorf("pipeline config is required")
}
if strings.EqualFold(strings.TrimSpace(cfg.Pipeline.Storage.Backend), "s3") {
if cfg.Pipeline.Storage.S3 == nil {
return nil, fmt.Errorf("pipeline.storage.s3 is required when pipeline.storage.backend is s3")
}
return NewS3BackendFromConfig(ctx, *cfg.Pipeline.Storage.S3)
}
if cfg.Pipeline.Storage.S3 != nil && strings.TrimSpace(cfg.Pipeline.Storage.S3.Bucket) != "" {
return NewS3BackendFromConfig(ctx, *cfg.Pipeline.Storage.S3)
}
return nil, fmt.Errorf("no remote object store backend is configured")
}

View File

@@ -0,0 +1,56 @@
package storage
import (
"context"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestNewObjectStoreFromConfigBuildsS3WhenBackendIsS3(t *testing.T) {
original := newS3Client
t.Cleanup(func() { newS3Client = original })
newS3Client = func(_ context.Context, _ s3ClientOptions) (s3API, error) {
return &fakeS3API{}, nil
}
store, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
Pipeline: &config.PipelineConfig{
Storage: config.StorageConfig{
Backend: "s3",
S3: &config.StorageS3Config{
Bucket: "my-archive",
},
},
},
})
if err != nil {
t.Fatalf("NewObjectStoreFromConfig() error = %v", err)
}
if _, ok := store.(*S3Backend); !ok {
t.Fatalf("store type = %T, want *S3Backend", store)
}
}
func TestNewObjectStoreFromConfigRequiresS3ConfigWhenBackendIsS3(t *testing.T) {
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
Pipeline: &config.PipelineConfig{
Storage: config.StorageConfig{Backend: "s3"},
},
})
if err == nil || !strings.Contains(err.Error(), "pipeline.storage.s3 is required") {
t.Fatalf("NewObjectStoreFromConfig() error = %v, want missing storage.s3 error", err)
}
}
func TestNewObjectStoreFromConfigNoRemoteBackendConfigured(t *testing.T) {
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
Pipeline: &config.PipelineConfig{
Storage: config.StorageConfig{Backend: "local"},
},
})
if err == nil || !strings.Contains(err.Error(), "no remote object store backend is configured") {
t.Fatalf("NewObjectStoreFromConfig() error = %v, want no-backend error", err)
}
}

View File

@@ -1,6 +1,14 @@
package storage
import "context"
import (
"context"
"fmt"
"os"
"path/filepath"
"sort"
"strings"
"time"
)
// NoopBackend is a deterministic no-op archive/storage adapter.
type NoopBackend struct{}
@@ -18,6 +26,21 @@ type FakeBackend struct {
Requests []ArchiveRequest
Err error
Result ArchiveResult
Objects map[string]FakeObject
Uploads []FakeUploadCall
ListErr error
DownloadErr error
UploadErr error
ExistsErr error
}
// FakeUploadCall captures one upload invocation in call order.
type FakeUploadCall struct {
LocalPath string
Key string
Options UploadOptions
}
// Archive records request and returns configured response.
@@ -38,3 +61,148 @@ func (f *FakeBackend) Archive(ctx context.Context, req ArchiveRequest) (ArchiveR
}
return res, nil
}
// FakeObject is a deterministic fake object-store record.
type FakeObject struct {
Key string
Data []byte
Metadata map[string]string
ETag string
LastModified *time.Time
}
// SeedObject inserts or replaces an object in the fake object store.
func (f *FakeBackend) SeedObject(obj FakeObject) {
if f.Objects == nil {
f.Objects = map[string]FakeObject{}
}
key := normalizeObjectKey(obj.Key)
obj.Key = key
obj.Data = append([]byte(nil), obj.Data...)
obj.Metadata = copyMetadata(obj.Metadata)
f.Objects[key] = obj
}
// List returns deterministic prefix-filtered objects.
func (f *FakeBackend) List(ctx context.Context, prefix string) ([]ObjectInfo, error) {
if err := ctx.Err(); err != nil {
return nil, err
}
if f.ListErr != nil {
return nil, f.ListErr
}
normalizedPrefix := normalizeObjectKey(prefix)
keys := make([]string, 0, len(f.Objects))
for key := range f.Objects {
if strings.HasPrefix(key, normalizedPrefix) {
keys = append(keys, key)
}
}
sort.Strings(keys)
out := make([]ObjectInfo, 0, len(keys))
for _, key := range keys {
obj := f.Objects[key]
out = append(out, ObjectInfo{
Key: obj.Key,
Size: int64(len(obj.Data)),
ETag: obj.ETag,
LastModified: obj.LastModified,
})
}
return out, nil
}
// Download writes one object to a local path.
func (f *FakeBackend) Download(ctx context.Context, key, localPath string) error {
if err := ctx.Err(); err != nil {
return err
}
if f.DownloadErr != nil {
return f.DownloadErr
}
if strings.TrimSpace(localPath) == "" {
return fmt.Errorf("download object: local path is required")
}
obj, ok := f.Objects[normalizeObjectKey(key)]
if !ok {
return fmt.Errorf("download object %q: %w", key, os.ErrNotExist)
}
if err := os.MkdirAll(filepath.Dir(localPath), 0o755); err != nil {
return fmt.Errorf("download object %q: create parent directory: %w", key, err)
}
if err := os.WriteFile(localPath, obj.Data, 0o644); err != nil {
return fmt.Errorf("download object %q: write local file: %w", key, err)
}
return nil
}
// Upload reads a local file and stores it under key.
func (f *FakeBackend) Upload(ctx context.Context, localPath, key string, opts UploadOptions) (ObjectInfo, error) {
if err := ctx.Err(); err != nil {
return ObjectInfo{}, err
}
if f.UploadErr != nil {
return ObjectInfo{}, f.UploadErr
}
if strings.TrimSpace(localPath) == "" {
return ObjectInfo{}, fmt.Errorf("upload object: local path is required")
}
if strings.TrimSpace(key) == "" {
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
}
data, err := os.ReadFile(localPath)
if err != nil {
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", key, localPath, err)
}
normalizedKey := normalizeObjectKey(key)
f.Uploads = append(f.Uploads, FakeUploadCall{
LocalPath: localPath,
Key: normalizedKey,
Options: UploadOptions{
Metadata: copyMetadata(opts.Metadata),
ContentType: opts.ContentType,
},
})
now := time.Now().UTC()
obj := FakeObject{
Key: normalizedKey,
Data: data,
Metadata: copyMetadata(opts.Metadata),
LastModified: &now,
}
f.SeedObject(obj)
return ObjectInfo{
Key: normalizedKey,
Size: int64(len(data)),
LastModified: &now,
}, nil
}
// Exists checks object presence.
func (f *FakeBackend) Exists(ctx context.Context, key string) (bool, error) {
if err := ctx.Err(); err != nil {
return false, err
}
if f.ExistsErr != nil {
return false, f.ExistsErr
}
_, ok := f.Objects[normalizeObjectKey(key)]
return ok, nil
}
func copyMetadata(in map[string]string) map[string]string {
if len(in) == 0 {
return nil
}
out := make(map[string]string, len(in))
for k, v := range in {
out[k] = v
}
return out
}

View File

@@ -3,6 +3,9 @@ package storage
import (
"context"
"errors"
"os"
"path/filepath"
"strings"
"testing"
)
@@ -29,3 +32,84 @@ func TestFakeBackendError(t *testing.T) {
t.Fatal("expected error, got nil")
}
}
func TestFakeBackendListPrefixFiltering(t *testing.T) {
fake := &FakeBackend{}
fake.SeedObject(FakeObject{Key: "dnd/campaigns/forsaken/audio/a.flac", Data: []byte("a")})
fake.SeedObject(FakeObject{Key: "dnd/campaigns/forsaken/audio/b.flac", Data: []byte("b")})
fake.SeedObject(FakeObject{Key: "dnd/campaigns/other/audio/c.flac", Data: []byte("c")})
items, err := fake.List(context.Background(), "dnd/campaigns/forsaken/audio/")
if err != nil {
t.Fatalf("List() error = %v", err)
}
if len(items) != 2 {
t.Fatalf("List() len = %d, want 2", len(items))
}
if items[0].Key != "dnd/campaigns/forsaken/audio/a.flac" || items[1].Key != "dnd/campaigns/forsaken/audio/b.flac" {
t.Fatalf("List() keys = %#v", items)
}
}
func TestFakeBackendDownload(t *testing.T) {
fake := &FakeBackend{}
fake.SeedObject(FakeObject{Key: "audio/a.flac", Data: []byte("audio-a")})
dst := filepath.Join(t.TempDir(), "nested", "a.flac")
if err := fake.Download(context.Background(), "audio/a.flac", dst); err != nil {
t.Fatalf("Download() error = %v", err)
}
data, err := os.ReadFile(dst)
if err != nil {
t.Fatalf("ReadFile() error = %v", err)
}
if string(data) != "audio-a" {
t.Fatalf("downloaded content = %q, want %q", string(data), "audio-a")
}
}
func TestFakeBackendUploadAndExists(t *testing.T) {
fake := &FakeBackend{}
local := filepath.Join(t.TempDir(), "upload.txt")
if err := os.WriteFile(local, []byte("payload"), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
info, err := fake.Upload(context.Background(), local, `runs\id\artifact.txt`, UploadOptions{
Metadata: map[string]string{"kind": "artifact"},
})
if err != nil {
t.Fatalf("Upload() error = %v", err)
}
if info.Key != "runs/id/artifact.txt" {
t.Fatalf("Upload() key = %q, want normalized key", info.Key)
}
ok, err := fake.Exists(context.Background(), "runs/id/artifact.txt")
if err != nil {
t.Fatalf("Exists() error = %v", err)
}
if !ok {
t.Fatal("Exists() = false, want true")
}
}
func TestFakeBackendObjectErrors(t *testing.T) {
fake := &FakeBackend{DownloadErr: errors.New("download fail"), UploadErr: errors.New("upload fail"), ListErr: errors.New("list fail"), ExistsErr: errors.New("exists fail")}
if _, err := fake.List(context.Background(), "x"); err == nil || !strings.Contains(err.Error(), "list fail") {
t.Fatalf("List() error = %v, want list fail", err)
}
if err := fake.Download(context.Background(), "x", filepath.Join(t.TempDir(), "x")); err == nil || !strings.Contains(err.Error(), "download fail") {
t.Fatalf("Download() error = %v, want download fail", err)
}
local := filepath.Join(t.TempDir(), "x.txt")
_ = os.WriteFile(local, []byte("x"), 0o644)
if _, err := fake.Upload(context.Background(), local, "x", UploadOptions{}); err == nil || !strings.Contains(err.Error(), "upload fail") {
t.Fatalf("Upload() error = %v, want upload fail", err)
}
if _, err := fake.Exists(context.Background(), "x"); err == nil || !strings.Contains(err.Error(), "exists fail") {
t.Fatalf("Exists() error = %v, want exists fail", err)
}
}

View File

@@ -0,0 +1,8 @@
package storage
import "strings"
func normalizeObjectKey(key string) string {
normalized := strings.ReplaceAll(strings.TrimSpace(key), "\\", "/")
return strings.TrimLeft(normalized, "/")
}

View File

@@ -0,0 +1,21 @@
package storage
import "testing"
func TestNormalizeObjectKey(t *testing.T) {
tests := []struct {
in string
want string
}{
{in: `dnd\campaigns\forsaken\a.flac`, want: "dnd/campaigns/forsaken/a.flac"},
{in: " /dnd/campaigns/forsaken/a.flac ", want: "dnd/campaigns/forsaken/a.flac"},
{in: "//dnd/campaigns/forsaken/a.flac", want: "dnd/campaigns/forsaken/a.flac"},
{in: "", want: ""},
}
for _, tt := range tests {
if got := normalizeObjectKey(tt.in); got != tt.want {
t.Fatalf("normalizeObjectKey(%q) = %q, want %q", tt.in, got, tt.want)
}
}
}

View File

@@ -0,0 +1,32 @@
package storage
import (
"context"
"time"
)
// ObjectStore is a remote object storage boundary used by future prepare/archive work.
//
// Key invariant:
// callers pass full bucket-relative object keys. Backend implementations do not
// infer Narratio session semantics and do not prepend root prefixes.
type ObjectStore interface {
List(ctx context.Context, prefix string) ([]ObjectInfo, error)
Download(ctx context.Context, key, localPath string) error
Upload(ctx context.Context, localPath, key string, opts UploadOptions) (ObjectInfo, error)
Exists(ctx context.Context, key string) (bool, error)
}
// ObjectInfo describes one object in remote storage.
type ObjectInfo struct {
Key string
Size int64
ETag string
LastModified *time.Time
}
// UploadOptions configures optional object upload metadata.
type UploadOptions struct {
Metadata map[string]string
ContentType string
}

View File

@@ -0,0 +1,275 @@
package storage
import (
"context"
"errors"
"fmt"
"io"
"os"
"path/filepath"
"strings"
"time"
awsconfig "github.com/aws/aws-sdk-go-v2/config"
"github.com/aws/aws-sdk-go-v2/credentials"
"github.com/aws/aws-sdk-go-v2/service/s3"
"github.com/aws/aws-sdk-go-v2/service/s3/types"
"github.com/aws/smithy-go"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
type s3API interface {
ListObjectsV2(ctx context.Context, params *s3.ListObjectsV2Input, optFns ...func(*s3.Options)) (*s3.ListObjectsV2Output, error)
GetObject(ctx context.Context, params *s3.GetObjectInput, optFns ...func(*s3.Options)) (*s3.GetObjectOutput, error)
PutObject(ctx context.Context, params *s3.PutObjectInput, optFns ...func(*s3.Options)) (*s3.PutObjectOutput, error)
HeadObject(ctx context.Context, params *s3.HeadObjectInput, optFns ...func(*s3.Options)) (*s3.HeadObjectOutput, error)
}
// S3Backend is an ObjectStore implementation backed by S3-compatible APIs.
type S3Backend struct {
bucket string
client s3API
}
type s3ClientOptions struct {
Region string
Endpoint string
ForcePathStyle bool
AccessKeyID string
SecretKey string
}
var newS3Client = func(ctx context.Context, opts s3ClientOptions) (s3API, error) {
loadOpts := make([]func(*awsconfig.LoadOptions) error, 0, 1)
if strings.TrimSpace(opts.Region) != "" {
loadOpts = append(loadOpts, awsconfig.WithRegion(strings.TrimSpace(opts.Region)))
}
if strings.TrimSpace(opts.AccessKeyID) != "" && strings.TrimSpace(opts.SecretKey) != "" {
loadOpts = append(loadOpts, awsconfig.WithCredentialsProvider(
credentials.NewStaticCredentialsProvider(
strings.TrimSpace(opts.AccessKeyID),
strings.TrimSpace(opts.SecretKey),
"",
),
))
}
awsCfg, err := awsconfig.LoadDefaultConfig(ctx, loadOpts...)
if err != nil {
return nil, fmt.Errorf("load aws config: %w", err)
}
return s3.NewFromConfig(awsCfg, func(o *s3.Options) {
if strings.TrimSpace(opts.Endpoint) != "" {
endpoint := strings.TrimSpace(opts.Endpoint)
o.BaseEndpoint = &endpoint
}
o.UsePathStyle = opts.ForcePathStyle
}), nil
}
// NewS3BackendFromConfig builds an S3 backend from resolved config.
func NewS3BackendFromConfig(ctx context.Context, cfg config.StorageS3Config) (*S3Backend, error) {
bucket := strings.TrimSpace(cfg.Bucket)
if bucket == "" {
return nil, fmt.Errorf("storage.s3.bucket is required")
}
client, err := newS3Client(ctx, s3ClientOptions{
Region: cfg.Region,
Endpoint: cfg.Endpoint,
ForcePathStyle: cfg.ForcePathStyle,
AccessKeyID: s3CredentialFromEnv(orDefaultEnvName(cfg.AccessKeyIDEnv, config.DefaultS3AccessKeyIDEnv)),
SecretKey: s3CredentialFromEnv(orDefaultEnvName(cfg.SecretKeyEnv, config.DefaultS3SecretAccessKeyEnv)),
})
if err != nil {
return nil, fmt.Errorf("build s3 client: %w", err)
}
return &S3Backend{
bucket: bucket,
client: client,
}, nil
}
func s3CredentialFromEnv(envVarName string) string {
name := strings.TrimSpace(envVarName)
if name == "" {
return ""
}
value, ok := os.LookupEnv(name)
if !ok {
return ""
}
return strings.TrimSpace(value)
}
func orDefaultEnvName(name, fallback string) string {
trimmed := strings.TrimSpace(name)
if trimmed == "" {
return fallback
}
return trimmed
}
// List returns objects under prefix.
func (b *S3Backend) List(ctx context.Context, prefix string) ([]ObjectInfo, error) {
normalizedPrefix := normalizeObjectKey(prefix)
out := make([]ObjectInfo, 0)
var token *string
for {
resp, err := b.client.ListObjectsV2(ctx, &s3.ListObjectsV2Input{
Bucket: &b.bucket,
Prefix: &normalizedPrefix,
ContinuationToken: token,
})
if err != nil {
return nil, fmt.Errorf("list objects under %q: %w", normalizedPrefix, err)
}
for _, item := range resp.Contents {
var lastModified *time.Time
if item.LastModified != nil {
t := *item.LastModified
lastModified = &t
}
out = append(out, ObjectInfo{
Key: normalizeObjectKey(valueOrEmpty(item.Key)),
Size: valueOrZeroInt64(item.Size),
ETag: strings.Trim(valueOrEmpty(item.ETag), "\""),
LastModified: lastModified,
})
}
if !valueOrFalseBool(resp.IsTruncated) || resp.NextContinuationToken == nil {
break
}
token = resp.NextContinuationToken
}
return out, nil
}
// Download retrieves one object to localPath, creating parent directories as needed.
func (b *S3Backend) Download(ctx context.Context, key, localPath string) error {
normalizedKey := normalizeObjectKey(key)
if strings.TrimSpace(localPath) == "" {
return fmt.Errorf("download object: local path is required")
}
resp, err := b.client.GetObject(ctx, &s3.GetObjectInput{
Bucket: &b.bucket,
Key: &normalizedKey,
})
if err != nil {
return fmt.Errorf("download object %q: %w", normalizedKey, err)
}
defer resp.Body.Close()
if err := os.MkdirAll(filepath.Dir(localPath), 0o755); err != nil {
return fmt.Errorf("download object %q: create parent directory: %w", normalizedKey, err)
}
dst, err := os.Create(localPath)
if err != nil {
return fmt.Errorf("download object %q: create local file: %w", normalizedKey, err)
}
defer dst.Close()
if _, err := io.Copy(dst, resp.Body); err != nil {
return fmt.Errorf("download object %q: copy body: %w", normalizedKey, err)
}
if err := dst.Sync(); err != nil {
return fmt.Errorf("download object %q: sync local file: %w", normalizedKey, err)
}
return nil
}
// Upload sends a local file to key.
func (b *S3Backend) Upload(ctx context.Context, localPath, key string, opts UploadOptions) (ObjectInfo, error) {
normalizedKey := normalizeObjectKey(key)
if strings.TrimSpace(localPath) == "" {
return ObjectInfo{}, fmt.Errorf("upload object: local path is required")
}
if normalizedKey == "" {
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
}
file, err := os.Open(localPath)
if err != nil {
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", normalizedKey, localPath, err)
}
defer file.Close()
stat, err := file.Stat()
if err != nil {
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: stat local file: %w", normalizedKey, localPath, err)
}
input := &s3.PutObjectInput{
Bucket: &b.bucket,
Key: &normalizedKey,
Body: file,
Metadata: copyMetadata(opts.Metadata),
}
if strings.TrimSpace(opts.ContentType) != "" {
ct := strings.TrimSpace(opts.ContentType)
input.ContentType = &ct
}
resp, err := b.client.PutObject(ctx, input)
if err != nil {
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", normalizedKey, localPath, err)
}
return ObjectInfo{
Key: normalizedKey,
Size: stat.Size(),
ETag: strings.Trim(valueOrEmpty(resp.ETag), "\""),
}, nil
}
// Exists checks whether one object key exists.
func (b *S3Backend) Exists(ctx context.Context, key string) (bool, error) {
normalizedKey := normalizeObjectKey(key)
_, err := b.client.HeadObject(ctx, &s3.HeadObjectInput{
Bucket: &b.bucket,
Key: &normalizedKey,
})
if err == nil {
return true, nil
}
var notFound *types.NotFound
if errors.As(err, &notFound) {
return false, nil
}
var apiErr smithy.APIError
if errors.As(err, &apiErr) {
switch apiErr.ErrorCode() {
case "NotFound", "NoSuchKey", "404":
return false, nil
}
}
return false, fmt.Errorf("head object %q: %w", normalizedKey, err)
}
func valueOrEmpty(v *string) string {
if v == nil {
return ""
}
return *v
}
func valueOrZeroInt64(v *int64) int64 {
if v == nil {
return 0
}
return *v
}
func valueOrFalseBool(v *bool) bool {
if v == nil {
return false
}
return *v
}

View File

@@ -0,0 +1,253 @@
package storage
import (
"context"
"io"
"os"
"path/filepath"
"strings"
"testing"
"time"
"github.com/aws/aws-sdk-go-v2/service/s3"
"github.com/aws/aws-sdk-go-v2/service/s3/types"
"github.com/aws/smithy-go"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
type fakeS3API struct {
listOut *s3.ListObjectsV2Output
listErr error
getBody io.ReadCloser
getErr error
putOut *s3.PutObjectOutput
putErr error
headErr error
lastList *s3.ListObjectsV2Input
lastGet *s3.GetObjectInput
lastPut *s3.PutObjectInput
lastHead *s3.HeadObjectInput
}
func (f *fakeS3API) ListObjectsV2(_ context.Context, params *s3.ListObjectsV2Input, _ ...func(*s3.Options)) (*s3.ListObjectsV2Output, error) {
f.lastList = params
if f.listErr != nil {
return nil, f.listErr
}
if f.listOut == nil {
return &s3.ListObjectsV2Output{}, nil
}
return f.listOut, nil
}
func (f *fakeS3API) GetObject(_ context.Context, params *s3.GetObjectInput, _ ...func(*s3.Options)) (*s3.GetObjectOutput, error) {
f.lastGet = params
if f.getErr != nil {
return nil, f.getErr
}
body := f.getBody
if body == nil {
body = io.NopCloser(strings.NewReader(""))
}
return &s3.GetObjectOutput{Body: body}, nil
}
func (f *fakeS3API) PutObject(_ context.Context, params *s3.PutObjectInput, _ ...func(*s3.Options)) (*s3.PutObjectOutput, error) {
f.lastPut = params
if f.putErr != nil {
return nil, f.putErr
}
if f.putOut == nil {
return &s3.PutObjectOutput{}, nil
}
return f.putOut, nil
}
func (f *fakeS3API) HeadObject(_ context.Context, params *s3.HeadObjectInput, _ ...func(*s3.Options)) (*s3.HeadObjectOutput, error) {
f.lastHead = params
if f.headErr != nil {
return nil, f.headErr
}
return &s3.HeadObjectOutput{}, nil
}
func TestS3BackendListAndKeyNormalization(t *testing.T) {
lastModified := time.Date(2026, 5, 16, 12, 0, 0, 0, time.UTC)
client := &fakeS3API{
listOut: &s3.ListObjectsV2Output{
Contents: []types.Object{
{Key: strPtr(`dnd\campaigns\forsaken\a.flac`), Size: int64Ptr(7), ETag: strPtr(`"abc"`), LastModified: &lastModified},
},
},
}
backend := &S3Backend{bucket: "bucket-1", client: client}
items, err := backend.List(context.Background(), `dnd\campaigns\`)
if err != nil {
t.Fatalf("List() error = %v", err)
}
if len(items) != 1 {
t.Fatalf("List() len = %d, want 1", len(items))
}
if items[0].Key != "dnd/campaigns/forsaken/a.flac" {
t.Fatalf("List() key = %q, want normalized slash key", items[0].Key)
}
if items[0].ETag != "abc" {
t.Fatalf("List() ETag = %q, want %q", items[0].ETag, "abc")
}
if client.lastList == nil || *client.lastList.Prefix != "dnd/campaigns/" {
t.Fatalf("List() prefix = %#v, want normalized prefix", client.lastList)
}
}
func TestS3BackendDownloadCreatesParentDirectory(t *testing.T) {
client := &fakeS3API{getBody: io.NopCloser(strings.NewReader("audio"))}
backend := &S3Backend{bucket: "bucket-1", client: client}
dst := filepath.Join(t.TempDir(), "nested", "clip.flac")
if err := backend.Download(context.Background(), `audio\clip.flac`, dst); err != nil {
t.Fatalf("Download() error = %v", err)
}
data, err := os.ReadFile(dst)
if err != nil {
t.Fatalf("ReadFile() error = %v", err)
}
if string(data) != "audio" {
t.Fatalf("downloaded content = %q, want %q", string(data), "audio")
}
if client.lastGet == nil || *client.lastGet.Key != "audio/clip.flac" {
t.Fatalf("GetObject key = %#v, want normalized key", client.lastGet)
}
}
func TestS3BackendUploadAndExists(t *testing.T) {
client := &fakeS3API{putOut: &s3.PutObjectOutput{ETag: strPtr(`"etag123"`)}}
backend := &S3Backend{bucket: "bucket-1", client: client}
local := filepath.Join(t.TempDir(), "artifact.txt")
if err := os.WriteFile(local, []byte("artifact"), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
info, err := backend.Upload(context.Background(), local, `runs\id\artifact.txt`, UploadOptions{
Metadata: map[string]string{"kind": "artifact"},
})
if err != nil {
t.Fatalf("Upload() error = %v", err)
}
if info.Key != "runs/id/artifact.txt" {
t.Fatalf("Upload key = %q, want normalized key", info.Key)
}
if info.ETag != "etag123" {
t.Fatalf("Upload ETag = %q, want %q", info.ETag, "etag123")
}
if client.lastPut == nil || *client.lastPut.Key != "runs/id/artifact.txt" {
t.Fatalf("PutObject key = %#v, want normalized key", client.lastPut)
}
ok, err := backend.Exists(context.Background(), "runs/id/artifact.txt")
if err != nil {
t.Fatalf("Exists() error = %v", err)
}
if !ok {
t.Fatal("Exists() = false, want true")
}
}
func TestS3BackendUploadMissingLocalFile(t *testing.T) {
backend := &S3Backend{bucket: "bucket-1", client: &fakeS3API{}}
_, err := backend.Upload(context.Background(), filepath.Join(t.TempDir(), "missing.txt"), "key.txt", UploadOptions{})
if err == nil || !strings.Contains(err.Error(), "no such file") {
t.Fatalf("Upload() error = %v, want missing local file error", err)
}
}
func TestS3BackendExistsNotFound(t *testing.T) {
backend := &S3Backend{
bucket: "bucket-1",
client: &fakeS3API{
headErr: &smithy.GenericAPIError{Code: "NotFound", Message: "missing"},
},
}
ok, err := backend.Exists(context.Background(), "missing-key")
if err != nil {
t.Fatalf("Exists() error = %v", err)
}
if ok {
t.Fatal("Exists() = true, want false")
}
}
func TestNewS3BackendFromConfigUsesClientOptions(t *testing.T) {
original := newS3Client
t.Cleanup(func() { newS3Client = original })
t.Setenv("OBJECT_STORAGE_KEY_ID", "id-123")
t.Setenv("OBJECT_STORAGE_KEY", "secret-abc")
var got s3ClientOptions
newS3Client = func(_ context.Context, opts s3ClientOptions) (s3API, error) {
got = opts
return &fakeS3API{}, nil
}
backend, err := NewS3BackendFromConfig(context.Background(), config.StorageS3Config{
Bucket: "my-archive",
Region: "us-east-1",
Endpoint: "http://localhost:9000",
ForcePathStyle: true,
})
if err != nil {
t.Fatalf("NewS3BackendFromConfig() error = %v", err)
}
if backend.bucket != "my-archive" {
t.Fatalf("backend.bucket = %q, want %q", backend.bucket, "my-archive")
}
if got.Region != "us-east-1" || got.Endpoint != "http://localhost:9000" || !got.ForcePathStyle {
t.Fatalf("client options = %#v, want region/endpoint/path-style values", got)
}
if got.AccessKeyID != "id-123" || got.SecretKey != "secret-abc" {
t.Fatalf("client options credentials = %#v, want env-resolved static credentials", got)
}
}
func TestNewS3BackendFromConfigRequiresBucket(t *testing.T) {
_, err := NewS3BackendFromConfig(context.Background(), config.StorageS3Config{})
if err == nil || !strings.Contains(err.Error(), "bucket is required") {
t.Fatalf("NewS3BackendFromConfig() error = %v, want bucket validation", err)
}
}
func TestNewS3BackendFromConfigFallsBackWhenCredentialEnvMissing(t *testing.T) {
original := newS3Client
t.Cleanup(func() { newS3Client = original })
var got s3ClientOptions
newS3Client = func(_ context.Context, opts s3ClientOptions) (s3API, error) {
got = opts
return &fakeS3API{}, nil
}
_, err := NewS3BackendFromConfig(context.Background(), config.StorageS3Config{
Bucket: "my-archive",
Region: "us-east-1",
AccessKeyIDEnv: "MISSING_ACCESS_KEY_ID",
SecretKeyEnv: "MISSING_SECRET_KEY",
})
if err != nil {
t.Fatalf("NewS3BackendFromConfig() error = %v", err)
}
if got.AccessKeyID != "" || got.SecretKey != "" {
t.Fatalf("client options credentials = %#v, want empty fallback values", got)
}
}
func strPtr(v string) *string { return &v }
func int64Ptr(v int64) *int64 { return &v }
var _ s3API = (*fakeS3API)(nil)

View File

@@ -0,0 +1,66 @@
package app
import (
"fmt"
"sort"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
type artifactSelectionFlag struct {
values []string
}
func (f *artifactSelectionFlag) String() string {
return strings.Join(f.values, ",")
}
func (f *artifactSelectionFlag) Set(value string) error {
f.values = append(f.values, value)
return nil
}
func (f *artifactSelectionFlag) Normalize() ([]string, error) {
if len(f.values) == 0 {
return nil, nil
}
seen := map[string]struct{}{}
out := make([]string, 0, len(f.values))
for _, raw := range f.values {
for _, part := range strings.Split(raw, ",") {
name := strings.TrimSpace(part)
if name == "" {
return nil, fmt.Errorf("artifact names must be non-empty")
}
if _, ok := seen[name]; ok {
continue
}
seen[name] = struct{}{}
out = append(out, name)
}
}
sort.Strings(out)
return out, nil
}
func validateSelectedAnalyzeArtifacts(cfg *config.Config, selected []string) error {
if len(selected) == 0 {
return nil
}
if cfg == nil || cfg.Pipeline == nil || cfg.Pipeline.Scriptorium == nil {
return fmt.Errorf("--artifacts requires pipeline.scriptorium.artifacts to be configured")
}
configured := cfg.Pipeline.Scriptorium.Artifacts
if len(configured) == 0 {
return fmt.Errorf("--artifacts requires at least one configured artifact in pipeline.scriptorium.artifacts")
}
for _, name := range selected {
if _, ok := configured[name]; !ok {
return fmt.Errorf("--artifacts includes unknown artifact %q", name)
}
}
return nil
}

View File

@@ -0,0 +1,140 @@
package app
import (
"bytes"
"context"
"os"
"path/filepath"
"strings"
"testing"
"time"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
func TestExecuteRunStageArtifactsNonAnalyzeFails(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "--artifacts", "session_recap", "polish"},
&stdout,
&stderr,
)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), `run-stage: --artifacts is only supported for stage "analyze"`) {
t.Fatalf("stderr = %q, want stage-gating error", stderr.String())
}
}
func TestExecuteUnknownArtifactsFailValidation(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute(
[]string{"run", "--config", pipelinePath, "--session", sessionPath, "--artifacts", "unknown_artifact"},
&stdout,
&stderr,
)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), `run: --artifacts includes unknown artifact "unknown_artifact"`) {
t.Fatalf("stderr = %q, want unknown-artifact validation error", stderr.String())
}
}
func TestRunStageArtifactsDoesNotImplyForce(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{}
seed := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
seed.MarkStageSucceeded("analyze", time.Date(2026, 5, 3, 10, 1, 0, 0, time.UTC), nil)
if err := store.Save(context.Background(), manifestPath, seed); err != nil {
t.Fatalf("save manifest: %v", err)
}
var out bytes.Buffer
err := RunStage(
context.Background(),
[]string{"--config", pipelinePath, "--session", sessionPath, "--artifacts", "session_recap,session_recap", "analyze"},
&out,
)
if err != nil {
t.Fatalf("RunStage() error = %v", err)
}
if !strings.Contains(out.String(), "stage=analyze executed=0 skipped=1 force=false") {
t.Fatalf("output = %q, want analyze skip without force", out.String())
}
}
func TestResumeArtifactsWithSucceededAnalyzeSkipsUnlessForced(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{}
seed := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
for _, stageName := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "analyze", "archive", "notify"} {
seed.MarkStageSucceeded(stageName, time.Date(2026, 5, 3, 10, 1, 0, 0, time.UTC), nil)
}
if err := store.Save(context.Background(), manifestPath, seed); err != nil {
t.Fatalf("save manifest: %v", err)
}
var out bytes.Buffer
err := Resume(
context.Background(),
[]string{"--config", pipelinePath, "--session", sessionPath, "--artifacts", "session_recap"},
&out,
)
if err != nil {
t.Fatalf("Resume() error = %v", err)
}
if !strings.Contains(out.String(), "has no remaining stages") {
t.Fatalf("output = %q, want no remaining stages", out.String())
}
}
func writeValidConfigFilesWithScriptoriumArtifacts(t *testing.T, workspaceRoot string) (string, string) {
t.Helper()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
f, err := os.OpenFile(pipelinePath, os.O_APPEND|os.O_WRONLY, 0)
if err != nil {
t.Fatalf("open pipeline config for append: %v", err)
}
defer f.Close()
extra := `
scriptorium:
binary: scriptorium
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
output_path: artifacts/session_recap.md
player_handout:
enabled: true
prompt_id: dnd.player_handout
output_path: artifacts/player_handout.md
depends_on:
- session_recap
inputs:
recap:
source: narratio.artifact.session_recap
required: true
`
if _, err := f.WriteString(extra); err != nil {
t.Fatalf("append scriptorium config: %v", err)
}
return pipelinePath, sessionPath
}

View File

@@ -0,0 +1,132 @@
package app
import (
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestArtifactSelectionFlagNormalize(t *testing.T) {
tests := []struct {
name string
inputs []string
want []string
wantErr string
}{
{
name: "single value",
inputs: []string{"session_recap"},
want: []string{"session_recap"},
},
{
name: "repeatable and comma separated values are deduped and sorted",
inputs: []string{"session_recap,player_handout", "session_recap"},
want: []string{"player_handout", "session_recap"},
},
{
name: "empty token fails",
inputs: []string{"session_recap,"},
wantErr: "artifact names must be non-empty",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
var flag artifactSelectionFlag
for _, in := range tt.inputs {
if err := flag.Set(in); err != nil {
t.Fatalf("Set(%q) error = %v", in, err)
}
}
got, err := flag.Normalize()
if tt.wantErr != "" {
if err == nil {
t.Fatalf("Normalize() error = nil, want %q", tt.wantErr)
}
if err.Error() != tt.wantErr {
t.Fatalf("Normalize() error = %q, want %q", err.Error(), tt.wantErr)
}
return
}
if err != nil {
t.Fatalf("Normalize() error = %v", err)
}
if len(got) != len(tt.want) {
t.Fatalf("Normalize() len = %d, want %d; got=%v", len(got), len(tt.want), got)
}
for i := range got {
if got[i] != tt.want[i] {
t.Fatalf("Normalize()[%d] = %q, want %q", i, got[i], tt.want[i])
}
}
})
}
}
func TestValidateSelectedAnalyzeArtifacts(t *testing.T) {
tests := []struct {
name string
cfg *config.Config
selected []string
wantErr string
}{
{
name: "empty selection is accepted",
cfg: &config.Config{},
selected: nil,
},
{
name: "scriptorium required when selected artifacts present",
cfg: &config.Config{Pipeline: &config.PipelineConfig{}},
selected: []string{"session_recap"},
wantErr: "--artifacts requires pipeline.scriptorium.artifacts to be configured",
},
{
name: "unknown selected artifact fails",
cfg: &config.Config{
Pipeline: &config.PipelineConfig{
Scriptorium: &config.ScriptoriumConfig{
Artifacts: map[string]config.ScriptoriumArtifactConfig{
"session_recap": {Enabled: true, PromptID: "dnd.session_recap", OutputPath: "artifacts/session_recap.md"},
},
},
},
},
selected: []string{"player_handout"},
wantErr: `--artifacts includes unknown artifact "player_handout"`,
},
{
name: "known selected artifacts are accepted",
cfg: &config.Config{
Pipeline: &config.PipelineConfig{
Scriptorium: &config.ScriptoriumConfig{
Artifacts: map[string]config.ScriptoriumArtifactConfig{
"session_recap": {Enabled: true, PromptID: "dnd.session_recap", OutputPath: "artifacts/session_recap.md"},
"player_handout": {Enabled: true, PromptID: "dnd.player_handout", OutputPath: "artifacts/player_handout.md"},
},
},
},
},
selected: []string{"player_handout", "session_recap"},
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
err := validateSelectedAnalyzeArtifacts(tt.cfg, tt.selected)
if tt.wantErr != "" {
if err == nil {
t.Fatalf("error = nil, want %q", tt.wantErr)
}
if err.Error() != tt.wantErr {
t.Fatalf("error = %q, want %q", err.Error(), tt.wantErr)
}
return
}
if err != nil {
t.Fatalf("error = %v, want nil", err)
}
})
}
}

View File

@@ -12,6 +12,7 @@ import (
"testing"
"time"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
@@ -63,12 +64,13 @@ func TestExecuteMissingRequiredFlags(t *testing.T) {
args []string
want string
}{
{name: "run missing flags", args: []string{"run"}, want: "run: --config and --session are required"},
{name: "plan missing flags", args: []string{"plan"}, want: "plan: --config and --session are required"},
{name: "run missing flags", args: []string{"run"}, want: "run: no pipeline config path provided and no default pipeline config found; searched:"},
{name: "plan missing flags", args: []string{"plan"}, want: "plan: no pipeline config path provided and no default pipeline config found; searched:"},
{name: "status missing flags", args: []string{"status"}, want: "status: --manifest is required"},
{name: "resume missing flags", args: []string{"resume"}, want: "resume: --config and --session are required"},
{name: "resume missing flags", args: []string{"resume"}, want: "resume: no pipeline config path provided and no default pipeline config found; searched:"},
{name: "run-stage missing name", args: []string{"run-stage", "--config", "a", "--session", "b"}, want: "run-stage: expected exactly one stage name"},
{name: "run-stage missing config flags", args: []string{"run-stage", "polish"}, want: "run-stage: --config and --session are required"},
{name: "run-stage missing config flags", args: []string{"run-stage", "polish"}, want: "run-stage: no pipeline config path provided and no default pipeline config found; searched:"},
{name: "run missing config uses defaults", args: []string{"run", "--session", "session.yml"}, want: "run: no pipeline config path provided and no default pipeline config found; searched:"},
}
for _, tc := range cases {
@@ -109,7 +111,7 @@ func TestExecuteRunStageUnknownFails(t *testing.T) {
func TestExecuteRunStageNormalizeIsAccepted(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, "https://example.com/transcribe")
workRoot := filepath.Join(workspaceRoot, "work", "2026-05-03")
workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "processed.json"), `{"segments":[{"id":1}]}`)
var stdout bytes.Buffer
@@ -154,7 +156,7 @@ func TestExecuteRunStageTranscribeUsesConfiguredWhisperXServer(t *testing.T) {
t.Fatal("expected whisperx server to be called at least once")
}
outPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "transcripts", "raw", "alice.json")
outPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "transcripts", "raw", "alice.json")
data, err := os.ReadFile(outPath)
if err != nil {
t.Fatalf("ReadFile(%q): %v", outPath, err)
@@ -168,6 +170,159 @@ func TestExecuteRunStageTranscribeUsesConfiguredWhisperXServer(t *testing.T) {
}
}
func TestExecuteRunStagePolishLoadsCredentialFromSecretsDir(t *testing.T) {
workspaceRoot := t.TempDir()
configDir := t.TempDir()
sessionID := "2026-05-03"
secretsDir := filepath.Join(configDir, "secrets")
if err := os.MkdirAll(secretsDir, 0o755); err != nil {
t.Fatalf("MkdirAll(%q): %v", secretsDir, err)
}
if err := os.WriteFile(filepath.Join(secretsDir, "OPENROUTER_API_KEY"), []byte("from-secret-file\n"), 0o600); err != nil {
t.Fatalf("write OPENROUTER_API_KEY secret file: %v", err)
}
seriatimBinary := writeSeriatimAppTestWrapper(t)
auditaBinary := writeAuditaAppTestWrapper(t)
t.Setenv("GO_WANT_APP_SERIATIM_HELPER", "1")
t.Setenv("GO_WANT_APP_AUDITA_HELPER", "1")
pipelinePath := filepath.Join(configDir, "pipeline.yml")
sessionPath := filepath.Join(configDir, "session.yml")
pipelineYAML := `workspace:
root: ` + workspaceRoot + `
storage:
backend: local
secrets:
env_dir: ./secrets
whisperx:
transcribe_url: https://example.com/transcribe
seriatim:
binary: ` + seriatimBinary + `
audita:
binary: ` + auditaBinary + `
llm_api_key_env: OPENROUTER_API_KEY
analyzer:
timeout: 20m
notification:
timeout: 10s
`
sessionYAML := `session_id: ` + sessionID + `
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
if err := os.WriteFile(pipelinePath, []byte(pipelineYAML), 0o644); err != nil {
t.Fatalf("write pipeline.yml: %v", err)
}
if err := os.WriteFile(sessionPath, []byte(sessionYAML), 0o644); err != nil {
t.Fatalf("write session.yml: %v", err)
}
originalWD, err := os.Getwd()
if err != nil {
t.Fatalf("Getwd(): %v", err)
}
if err := os.Chdir(configDir); err != nil {
t.Fatalf("Chdir(%q): %v", configDir, err)
}
t.Cleanup(func() {
_ = os.Chdir(originalWD)
})
workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", sessionID)
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "merged.json"), `{"schema":"seriatim-intermediate","segments":[]}`)
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "[]\n")
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "--force", "polish"}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "stage=polish executed=1 skipped=0") {
t.Fatalf("stdout = %q, want polish execution", stdout.String())
}
}
func TestExecuteRunFailsWhenConfiguredSecretsDirMissing(t *testing.T) {
workspaceRoot := t.TempDir()
configDir := t.TempDir()
pipelinePath := filepath.Join(configDir, "pipeline.yml")
sessionPath := filepath.Join(configDir, "session.yml")
pipelineYAML := `workspace:
root: ` + workspaceRoot + `
storage:
backend: local
secrets:
env_dir: ./missing-secrets
whisperx:
transcribe_url: https://example.com/transcribe
seriatim:
binary: seriatim
audita:
binary: audita
analyzer:
timeout: 20m
notification:
timeout: 10s
`
sessionYAML := `session_id: 2026-05-03
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
if err := os.WriteFile(pipelinePath, []byte(pipelineYAML), 0o644); err != nil {
t.Fatalf("write pipeline.yml: %v", err)
}
if err := os.WriteFile(sessionPath, []byte(sessionYAML), 0o644); err != nil {
t.Fatalf("write session.yml: %v", err)
}
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"run", "--config", pipelinePath, "--session", sessionPath}, &stdout, &stderr)
if code == 0 {
t.Fatal("exit code = 0, want non-zero")
}
if !strings.Contains(stderr.String(), "read secrets env_dir") {
t.Fatalf("stderr = %q, want secrets read-dir error context", stderr.String())
}
}
func TestExecuteUsesDefaultPipelineConfigPathWhenConfigFlagOmitted(t *testing.T) {
workspaceRoot := t.TempDir()
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.Header().Set("Content-Type", "application/json")
_, _ = w.Write([]byte(`{"source":"default-config-test","segments":[{"speaker":"alice"}]}`))
}))
defer srv.Close()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL)
originalDefaults := append([]string(nil), config.DefaultPipelineConfigSearchPaths...)
config.DefaultPipelineConfigSearchPaths = []string{pipelinePath}
defer func() {
config.DefaultPipelineConfigSearchPaths = originalDefaults
}()
var stdout bytes.Buffer
var stderr bytes.Buffer
code := Execute([]string{"run", "--session", sessionPath}, &stdout, &stderr)
if code != 0 {
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
}
if !strings.Contains(stdout.String(), "narratio run: session 2026-05-03; executed=9 skipped=0; manifest=") {
t.Fatalf("stdout = %q, want successful run output", stdout.String())
}
}
func TestExecuteInvalidCommand(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
@@ -224,6 +379,11 @@ func writeValidConfigFiles(t *testing.T, workspaceRoot string, transcribeURL ...
root: ` + workspaceRoot + `
storage:
backend: s3
s3:
bucket: test-bucket
archive:
enabled: true
upload_run: false
whisperx:
transcribe_url: ` + url + `
timeout: 2s
@@ -247,6 +407,7 @@ notification:
`
sessionYAML := `session_id: 2026-05-03
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml

View File

@@ -0,0 +1,49 @@
package app
import (
"errors"
"fmt"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func resolvePipelineConfigPath(flagValue string) (string, error) {
return resolvePipelineConfigPathWithCandidates(flagValue, config.DefaultPipelineConfigSearchPaths)
}
func resolvePipelineConfigPathWithCandidates(flagValue string, candidates []string) (string, error) {
if explicit := strings.TrimSpace(flagValue); explicit != "" {
return explicit, nil
}
ordered := make([]string, 0, len(candidates))
for _, raw := range candidates {
path := strings.TrimSpace(raw)
if path == "" {
continue
}
ordered = append(ordered, path)
info, err := os.Stat(path)
if err == nil {
if info.IsDir() {
continue
}
return filepath.Clean(path), nil
}
if errors.Is(err, os.ErrNotExist) {
continue
}
return "", fmt.Errorf("check default pipeline config %q: %w", path, err)
}
if len(ordered) == 0 {
return "", fmt.Errorf("no pipeline config path provided and no default locations configured")
}
return "", fmt.Errorf(
"no pipeline config path provided and no default pipeline config found; searched: %s",
strings.Join(ordered, ", "),
)
}

View File

@@ -0,0 +1,65 @@
package app
import (
"os"
"path/filepath"
"strings"
"testing"
)
func TestResolvePipelineConfigPathWithCandidatesExplicitWins(t *testing.T) {
got, err := resolvePipelineConfigPathWithCandidates(" ./custom/pipeline.yml ", []string{"/a", "/b"})
if err != nil {
t.Fatalf("resolvePipelineConfigPathWithCandidates() error = %v", err)
}
if got != "./custom/pipeline.yml" {
t.Fatalf("resolved path = %q, want explicit path", got)
}
}
func TestResolvePipelineConfigPathWithCandidatesUsesFirstExisting(t *testing.T) {
dir := t.TempDir()
first := filepath.Join(dir, "first.yml")
second := filepath.Join(dir, "second.yml")
if err := os.WriteFile(second, []byte("workspace:\n root: ./tmp\n"), 0o644); err != nil {
t.Fatalf("write second default: %v", err)
}
got, err := resolvePipelineConfigPathWithCandidates("", []string{first, second})
if err != nil {
t.Fatalf("resolvePipelineConfigPathWithCandidates() error = %v", err)
}
if got != filepath.Clean(second) {
t.Fatalf("resolved path = %q, want %q", got, filepath.Clean(second))
}
}
func TestResolvePipelineConfigPathWithCandidatesPrecedence(t *testing.T) {
dir := t.TempDir()
first := filepath.Join(dir, "first.yml")
second := filepath.Join(dir, "second.yml")
if err := os.WriteFile(first, []byte("workspace:\n root: ./tmp\n"), 0o644); err != nil {
t.Fatalf("write first default: %v", err)
}
if err := os.WriteFile(second, []byte("workspace:\n root: ./tmp\n"), 0o644); err != nil {
t.Fatalf("write second default: %v", err)
}
got, err := resolvePipelineConfigPathWithCandidates("", []string{first, second})
if err != nil {
t.Fatalf("resolvePipelineConfigPathWithCandidates() error = %v", err)
}
if got != filepath.Clean(first) {
t.Fatalf("resolved path = %q, want first candidate %q", got, filepath.Clean(first))
}
}
func TestResolvePipelineConfigPathWithCandidatesMissing(t *testing.T) {
_, err := resolvePipelineConfigPathWithCandidates("", []string{"/does/not/exist/one.yml", "/does/not/exist/two.yml"})
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "no default pipeline config found") {
t.Fatalf("error = %q, want missing-defaults context", err.Error())
}
}

View File

@@ -5,9 +5,12 @@ import (
"flag"
"fmt"
"io"
"log/slog"
"os"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/logging"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
@@ -18,9 +21,11 @@ func Plan(ctx context.Context, args []string, out io.Writer) error {
var pipelinePath string
var sessionPath string
var sessionID string
var force bool
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml")
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
fs.StringVar(&sessionPath, "session", "", "path to session.yml")
fs.StringVar(&sessionID, "session-id", "", "session identifier for session.yml templates")
fs.BoolVar(&force, "force", false, "force stage execution (reserved for future behavior)")
if err := fs.Parse(args); err != nil {
@@ -29,20 +34,30 @@ func Plan(ctx context.Context, args []string, out io.Writer) error {
if fs.NArg() != 0 {
return fmt.Errorf("plan: unexpected positional arguments")
}
if pipelinePath == "" || sessionPath == "" {
return fmt.Errorf("plan: --config and --session are required")
resolvedPipelinePath, err := resolvePipelineConfigPath(pipelinePath)
if err != nil {
return fmt.Errorf("plan: %w", err)
}
resolvedSessionPath, err := resolveSessionConfigPath(sessionPath)
if err != nil {
return fmt.Errorf("plan: %w", err)
}
cfg, err := config.Load(pipelinePath, sessionPath)
cfg, err := config.LoadWithSessionOptions(resolvedPipelinePath, resolvedSessionPath, config.SessionLoadOptions{
SessionID: sessionID,
})
if err != nil {
return fmt.Errorf("plan: %w", err)
}
if err := config.Validate(cfg); err != nil {
return fmt.Errorf("plan: %w", err)
}
if _, err := loadSecretsFromConfig(cfg, logging.NewLogger(os.Stderr, slog.LevelInfo)); err != nil {
return fmt.Errorf("plan: %w", err)
}
store := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root)
paths, err := store.EnsureLayout(cfg.Session.SessionID)
paths, err := store.EnsureLayoutFor(cfg.Session.Campaign, cfg.Session.SessionID)
if err != nil {
return fmt.Errorf("plan: prepare workdir: %w", err)
}

View File

@@ -36,7 +36,7 @@ func TestPlanCreatesAndReusesWorkdir(t *testing.T) {
t.Fatalf("first output = %q, want totals", got)
}
sessionWorkdir := artifacts.SessionWorkDir(workspaceRoot, "2026-05-03")
sessionWorkdir := artifacts.SessionWorkDirForCampaign(workspaceRoot, "sample-campaign", "2026-05-03")
expectedDirs := []string{
sessionWorkdir,
filepath.Join(sessionWorkdir, "inputs"),
@@ -63,7 +63,7 @@ func TestPlanCreatesAndReusesWorkdir(t *testing.T) {
func TestPlanShowsRunAndSkipFromManifest(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json")
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{}
m := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
@@ -89,6 +89,54 @@ func TestPlanShowsRunAndSkipFromManifest(t *testing.T) {
}
}
func TestPlanFailsWhenConfiguredSecretsDirMissing(t *testing.T) {
workspaceRoot := t.TempDir()
configDir := t.TempDir()
pipelinePath := filepath.Join(configDir, "pipeline.yml")
sessionPath := filepath.Join(configDir, "session.yml")
pipelineYAML := `workspace:
root: ` + workspaceRoot + `
storage:
backend: local
secrets:
env_dir: ./missing-secrets
whisperx:
transcribe_url: https://example.com/transcribe
seriatim:
binary: seriatim
audita:
binary: audita
analyzer:
timeout: 20m
notification:
timeout: 10s
`
sessionYAML := `session_id: 2026-05-03
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
if err := os.WriteFile(pipelinePath, []byte(pipelineYAML), 0o644); err != nil {
t.Fatalf("write pipeline.yml: %v", err)
}
if err := os.WriteFile(sessionPath, []byte(sessionYAML), 0o644); err != nil {
t.Fatalf("write session.yml: %v", err)
}
var out bytes.Buffer
err := Plan(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath}, &out)
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "read secrets env_dir") {
t.Fatalf("error = %q, want secrets read error context", err.Error())
}
}
func assertDir(t *testing.T, path string) {
t.Helper()
info, err := os.Stat(path)

View File

@@ -0,0 +1,208 @@
package app
import (
"context"
"fmt"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
func runPostArchiveCleanup(ctx context.Context, env *Env, manifestPath string, m *manifest.Manifest, executed []string) error {
if env == nil || env.Config == nil || env.Config.Pipeline == nil || m == nil {
return nil
}
spoolRequested := env.Config.Pipeline.Spool.DeleteAudioAfterArchive
workRequested := env.Config.Pipeline.Workspace.CleanupAfterArchive
if !spoolRequested && !workRequested {
return nil
}
sr := archiveStageRecordForCleanup(m, executed)
if sr == nil {
return nil
}
if sr.Metadata == nil {
sr.Metadata = map[string]any{}
}
sr.Metadata["spool_cleanup_requested"] = spoolRequested
sr.Metadata["workdir_cleanup_requested"] = workRequested
eligible, reason := archiveCleanupEligible(env.Config, sr)
if !eligible {
sr.Metadata["cleanup_skipped"] = true
sr.Metadata["cleanup_skipped_reason"] = reason
if err := env.ManifestStore.Save(ctx, manifestPath, m); err != nil {
return fmt.Errorf("save manifest cleanup skip metadata %q: %w", manifestPath, err)
}
return nil
}
spoolDir := strings.TrimSpace(m.LocalSpoolDir)
if spoolDir == "" {
spoolDir = artifacts.SessionSpoolAudioDir(
env.Config.Pipeline.Spool.Root,
strings.TrimSpace(env.Config.Session.Campaign),
strings.TrimSpace(env.Config.Session.SessionID),
strings.TrimSpace(m.RunID),
)
}
workDir := strings.TrimSpace(m.LocalWorkDir)
if workDir == "" {
workDir = artifacts.SessionRunRootForCampaign(
env.Config.Pipeline.Workspace.Root,
strings.TrimSpace(env.Config.Session.Campaign),
strings.TrimSpace(env.Config.Session.SessionID),
strings.TrimSpace(m.RunID),
)
}
if spoolRequested {
if err := removeRunScopedDir(strings.TrimSpace(env.Config.Pipeline.Spool.Root), spoolDir, "pipeline.spool.delete_audio_after_archive"); err != nil {
sr.Metadata["cleanup_failed"] = true
sr.Metadata["cleanup_failed_policy"] = "pipeline.spool.delete_audio_after_archive"
sr.Metadata["cleanup_failed_path"] = spoolDir
_ = env.ManifestStore.Save(ctx, manifestPath, m)
return err
}
sr.Metadata["spool_cleanup_deleted"] = filepath.Clean(spoolDir)
}
if !workRequested {
sr.Metadata["cleanup_completed"] = true
sr.Metadata["cleanup_skipped"] = false
if err := env.ManifestStore.Save(ctx, manifestPath, m); err != nil {
return fmt.Errorf("save manifest cleanup metadata %q: %w", manifestPath, err)
}
return nil
}
if err := removeRunScopedDir(strings.TrimSpace(env.Config.Pipeline.Workspace.Root), workDir, "pipeline.workspace.cleanup_after_archive"); err != nil {
sr.Metadata["cleanup_failed"] = true
sr.Metadata["cleanup_failed_policy"] = "pipeline.workspace.cleanup_after_archive"
sr.Metadata["cleanup_failed_path"] = workDir
_ = env.ManifestStore.Save(ctx, manifestPath, m)
return err
}
sr.Metadata["workdir_cleanup_deleted"] = filepath.Clean(workDir)
sr.Metadata["cleanup_completed"] = true
sr.Metadata["cleanup_skipped"] = false
return nil
}
func archiveStageRecordForCleanup(m *manifest.Manifest, executed []string) *manifest.StageRecord {
if m == nil {
return nil
}
archiveRan := false
for _, name := range executed {
if name == "archive" {
archiveRan = true
break
}
}
if !archiveRan {
return nil
}
sr := m.Stages["archive"]
if sr == nil || sr.Status != manifest.StatusSucceeded {
return nil
}
return sr
}
func archiveCleanupEligible(cfg *config.Config, sr *manifest.StageRecord) (bool, string) {
if cfg == nil || cfg.Pipeline == nil || cfg.Pipeline.Archive == nil {
return false, "archive configuration is missing"
}
enabled := true
if cfg.Pipeline.Archive.Enabled != nil {
enabled = *cfg.Pipeline.Archive.Enabled
}
if !enabled {
return false, "archive.enabled is false"
}
uploadRun := true
if cfg.Pipeline.Archive.UploadRun != nil {
uploadRun = *cfg.Pipeline.Archive.UploadRun
}
if !uploadRun {
return false, "archive.upload_run is false"
}
if sr == nil || sr.Metadata == nil {
return false, "archive metadata is missing"
}
if skipped, _ := sr.Metadata["skipped"].(bool); skipped {
return false, "archive stage was skipped"
}
if uploaded, _ := sr.Metadata["uploaded"].(bool); !uploaded {
return false, "archive did not upload run record"
}
if pointer, _ := sr.Metadata["current_pointer_written"].(bool); !pointer {
return false, "archive did not write current pointer"
}
if strings.TrimSpace(asString(sr.Metadata["current_run_id_key"])) == "" {
return false, "archive current run pointer key is missing"
}
return true, ""
}
func removeRunScopedDir(root, target, policy string) error {
cleanRoot := strings.TrimSpace(root)
cleanTarget := strings.TrimSpace(target)
if cleanRoot == "" {
return fmt.Errorf("cleanup policy %s: root path is required", policy)
}
if cleanTarget == "" {
return fmt.Errorf("cleanup policy %s: target path is required", policy)
}
rootAbs, err := filepath.Abs(cleanRoot)
if err != nil {
return fmt.Errorf("cleanup policy %s: resolve root %q: %w", policy, cleanRoot, err)
}
targetAbs, err := filepath.Abs(cleanTarget)
if err != nil {
return fmt.Errorf("cleanup policy %s: resolve target %q: %w", policy, cleanTarget, err)
}
rel, err := filepath.Rel(rootAbs, targetAbs)
if err != nil {
return fmt.Errorf("cleanup policy %s: relative path from %q to %q: %w", policy, rootAbs, targetAbs, err)
}
if rel == "." {
return fmt.Errorf("cleanup policy %s: refusing to delete root directory %q", policy, rootAbs)
}
if rel == ".." || strings.HasPrefix(rel, ".."+string(filepath.Separator)) {
return fmt.Errorf("cleanup policy %s: refusing to delete path outside root: root=%q target=%q", policy, rootAbs, targetAbs)
}
info, err := os.Lstat(targetAbs)
if err != nil {
if os.IsNotExist(err) {
return nil
}
return fmt.Errorf("cleanup policy %s: stat target %q: %w", policy, targetAbs, err)
}
if info.Mode()&os.ModeSymlink != 0 {
return fmt.Errorf("cleanup policy %s: refusing to delete symlink path %q", policy, targetAbs)
}
if !info.IsDir() {
return fmt.Errorf("cleanup policy %s: target %q is not a directory", policy, targetAbs)
}
if err := os.RemoveAll(targetAbs); err != nil {
return fmt.Errorf("cleanup policy %s: remove %q: %w", policy, targetAbs, err)
}
return nil
}
func asString(v any) string {
s, _ := v.(string)
return s
}

View File

@@ -0,0 +1,403 @@
package app
import (
"context"
"errors"
"os"
"path/filepath"
"strings"
"testing"
"time"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
"gitea.maximumdirect.net/eric/narratio/internal/stage"
)
type archiveSuccessStage struct {
metadata map[string]any
}
func (archiveSuccessStage) Name() string { return "archive" }
func (archiveSuccessStage) Declares() stage.IODecl { return stage.IODecl{} }
func (s archiveSuccessStage) Run(_ context.Context, _ *stage.Env, _ *manifest.Manifest) (*stage.StageResult, error) {
md := map[string]any{
"stage": "archive",
"uploaded": true,
"current_pointer_written": true,
"current_run_id_key": "dnd/campaigns/sample-campaign/sessions/2026-05-03/current/run_id.txt",
}
for k, v := range s.metadata {
md[k] = v
}
return &stage.StageResult{Metadata: md}, nil
}
type notifyFailStage struct{}
func (notifyFailStage) Name() string { return "notify" }
func (notifyFailStage) Declares() stage.IODecl { return stage.IODecl{} }
func (notifyFailStage) Run(_ context.Context, _ *stage.Env, _ *manifest.Manifest) (*stage.StageResult, error) {
return nil, errors.New("notify failed")
}
func TestPostArchiveCleanupDisabledKeepsLocalDirs(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = false
cfg.Pipeline.Workspace.CleanupAfterArchive = false
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
assertExists(t, seed.localSourceAudio)
}
func TestPostArchiveCleanupSpoolOnly(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = false
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertMissing(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
assertExists(t, seed.localSourceAudio)
}
func TestPostArchiveCleanupWorkdirOnly(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = false
cfg.Pipeline.Workspace.CleanupAfterArchive = true
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertExists(t, cfg.Pipeline.Workspace.Root)
assertExists(t, seed.otherRunDir)
assertMissing(t, seed.runWorkDir)
assertExists(t, seed.spoolAudioDir)
}
func TestPostArchiveCleanupBothPolicies(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertMissing(t, seed.spoolAudioDir)
assertMissing(t, seed.runWorkDir)
assertExists(t, seed.otherRunDir)
}
func TestPostArchiveCleanupNotRunWhenArchiveFails(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
_, err := executeStages(context.Background(), cfg, []stage.Stage{failingStage{name: "archive", err: errors.New("archive failed")}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}})
if err == nil || !strings.Contains(err.Error(), "stage \"archive\" failed") {
t.Fatalf("executeStages() error = %v, want archive failure", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
func TestPostArchiveCleanupNotRunWhenArchiveSkipped(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{metadata: map[string]any{"skipped": true}}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
func TestPostArchiveCleanupNotRunWhenCurrentPointerMissing(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{metadata: map[string]any{"current_pointer_written": false}}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
func TestPostArchiveCleanupNotRunWhenArchiveUploadDisabled(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
cfg.Pipeline.Archive.UploadRun = boolPtr(false)
if _, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}}); err != nil {
t.Fatalf("executeStages() error = %v", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
func TestPostArchiveCleanupWaitsUntilAllStagesSucceed(t *testing.T) {
cfg, seed := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
_, err := executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}, notifyFailStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}})
if err == nil || !strings.Contains(err.Error(), "stage \"notify\" failed") {
t.Fatalf("executeStages() error = %v, want notify failure", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
func TestPostArchiveCleanupFailsOnUnsafePath(t *testing.T) {
cfg, _ := cleanupFixtureConfig(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = false
manifestPath := manifestPathFor(cfg)
store := &manifest.LocalStore{}
m, err := store.Load(context.Background(), manifestPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
m.LocalSpoolDir = filepath.Join(filepath.Dir(cfg.Pipeline.Spool.Root), "outside-spool")
if err := store.Save(context.Background(), manifestPath, m); err != nil {
t.Fatalf("Save() error = %v", err)
}
_, err = executeStages(context.Background(), cfg, []stage.Stage{archiveSuccessStage{}}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}})
if err == nil || !strings.Contains(err.Error(), "refusing to delete path outside root") {
t.Fatalf("executeStages() error = %v, want safe-path failure", err)
}
}
func TestPostArchiveCleanupNotRunWhenPromotionIsMissing(t *testing.T) {
cfg, seed, runID := archiveStageCleanupFixture(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
cfg.Pipeline.Archive.PromoteArtifacts = []config.ArchivePromotionRule{
{From: "artifacts/missing.md", To: "artifacts/missing.md", Required: boolPtr(true)},
}
archiveStageImpl, err := stage.Select("archive")
if err != nil {
t.Fatalf("Select(archive) error = %v", err)
}
_, err = executeStages(context.Background(), cfg, []stage.Stage{archiveStageImpl}, RunOptions{Env: &Env{ObjectStore: &storage.FakeBackend{}}})
if err == nil || !strings.Contains(err.Error(), "required promotion source missing") {
t.Fatalf("executeStages() error = %v, want promotion-missing failure", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
assertExists(t, filepath.Join(seed.runWorkDir, "manifest.json"))
assertExists(t, artifacts.SessionRunRootForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID, runID))
}
func TestPostArchiveCleanupNotRunWhenCurrentManifestUploadFails(t *testing.T) {
cfg, seed, _ := archiveStageCleanupFixture(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
failKey := seed.sessionPrefix + "current/manifest.json"
archiveStageImpl, err := stage.Select("archive")
if err != nil {
t.Fatalf("Select(archive) error = %v", err)
}
_, err = executeStages(context.Background(), cfg, []stage.Stage{archiveStageImpl}, RunOptions{
Env: &Env{ObjectStore: &failKeyStore{delegate: &storage.FakeBackend{}, failKey: failKey}},
})
if err == nil || !strings.Contains(err.Error(), "current manifest") {
t.Fatalf("executeStages() error = %v, want current-manifest failure", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
func TestPostArchiveCleanupNotRunWhenCurrentPointerUploadFails(t *testing.T) {
cfg, seed, _ := archiveStageCleanupFixture(t)
cfg.Pipeline.Spool.DeleteAudioAfterArchive = true
cfg.Pipeline.Workspace.CleanupAfterArchive = true
failKey := seed.sessionPrefix + "current/run_id.txt"
archiveStageImpl, err := stage.Select("archive")
if err != nil {
t.Fatalf("Select(archive) error = %v", err)
}
_, err = executeStages(context.Background(), cfg, []stage.Stage{archiveStageImpl}, RunOptions{
Env: &Env{ObjectStore: &failKeyStore{delegate: &storage.FakeBackend{}, failKey: failKey}},
})
if err == nil || !strings.Contains(err.Error(), "current run pointer") {
t.Fatalf("executeStages() error = %v, want current-run-pointer failure", err)
}
assertExists(t, seed.spoolAudioDir)
assertExists(t, seed.runWorkDir)
}
type cleanupSeed struct {
runWorkDir string
otherRunDir string
spoolAudioDir string
localSourceAudio string
sessionPrefix string
}
func cleanupFixtureConfig(t *testing.T) (*config.Config, cleanupSeed) {
t.Helper()
cfg := testConfig(t)
cfg.Pipeline.Archive = &config.ArchiveConfig{Enabled: boolPtr(true), UploadRun: boolPtr(true)}
cfg.Pipeline.Spool.Root = filepath.Join(t.TempDir(), "spool")
runID := "20260516T010203Z-1a2b3c4d"
runWorkDir := artifacts.SessionRunRootForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID, runID)
otherRunDir := artifacts.SessionRunRootForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID, "20260516T010204Z-5e6f7a8b")
spoolAudioDir := artifacts.SessionSpoolAudioDir(cfg.Pipeline.Spool.Root, cfg.Session.Campaign, cfg.Session.SessionID, runID)
mustWriteFile(t, filepath.Join(runWorkDir, "manifest.json"), "{}\n")
mustWriteFile(t, filepath.Join(runWorkDir, "logs", "stage.log"), "log\n")
mustWriteFile(t, filepath.Join(otherRunDir, "logs", "stage.log"), "other\n")
mustWriteFile(t, filepath.Join(spoolAudioDir, "speaker.flac"), "flac\n")
localSourceAudio := filepath.Join(filepath.Dir(cfg.SessionPath), "audio", "alice.flac")
mustWriteFile(t, localSourceAudio, "source\n")
seed := manifest.New(cfg.Session.SessionID, time.Now().UTC())
seed.Campaign = cfg.Session.Campaign
seed.RunID = runID
seed.LocalWorkDir = runWorkDir
seed.LocalSpoolDir = spoolAudioDir
seed.S3Bucket = "my-dnd-archive"
seed.S3SessionPrefix = "dnd/campaigns/sample-campaign/sessions/2026-05-03/"
seed.S3RunPrefix = seed.S3SessionPrefix + "runs/" + runID + "/"
store := &manifest.LocalStore{}
if err := os.MkdirAll(filepath.Dir(manifestPathFor(cfg)), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
if err := store.Save(context.Background(), manifestPathFor(cfg), seed); err != nil {
t.Fatalf("seed manifest save error = %v", err)
}
return cfg, cleanupSeed{
runWorkDir: runWorkDir,
otherRunDir: otherRunDir,
spoolAudioDir: spoolAudioDir,
localSourceAudio: localSourceAudio,
sessionPrefix: seed.S3SessionPrefix,
}
}
func archiveStageCleanupFixture(t *testing.T) (*config.Config, cleanupSeed, string) {
t.Helper()
cfg, seed := cleanupFixtureConfig(t)
runID := "20260516T010203Z-1a2b3c4d"
cfg.Pipeline.Storage.S3 = &config.StorageS3Config{
Bucket: "my-dnd-archive",
RootPrefix: "dnd",
}
cfg.Pipeline.Archive = &config.ArchiveConfig{
Enabled: boolPtr(true),
UploadRun: boolPtr(true),
PromoteArtifacts: []config.ArchivePromotionRule{
{From: "transcripts/trimmed.json", To: "transcripts/trimmed.json", Required: boolPtr(true)},
{From: "artifacts/session_recap.md", To: "artifacts/session_recap.md", Required: boolPtr(true)},
},
}
writeArchiveFixtureRunFiles(
t,
seed.runWorkDir,
artifacts.SessionWorkDirForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID),
)
store := &manifest.LocalStore{}
seedManifest, err := store.Load(context.Background(), manifestPathFor(cfg))
if err != nil {
t.Fatalf("Load() error = %v", err)
}
for _, name := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "analyze"} {
seedManifest.MarkStageSucceeded(name, time.Now().UTC(), nil)
}
seedManifest.S3SessionPrefix = artifacts.S3SessionPrefix("dnd", cfg.Session.Campaign, cfg.Session.SessionID)
seedManifest.S3RunPrefix = artifacts.S3RunPrefix(seedManifest.S3SessionPrefix, runID)
if err := store.Save(context.Background(), manifestPathFor(cfg), seedManifest); err != nil {
t.Fatalf("Save() error = %v", err)
}
seed.sessionPrefix = seedManifest.S3SessionPrefix
return cfg, seed, runID
}
func writeArchiveFixtureRunFiles(t *testing.T, runWorkDir, sessionRoot string) {
t.Helper()
mustWriteFile(t, filepath.Join(runWorkDir, "prepare", "inputs", "session.yml"), "session_id: 2026-05-03\n")
mustWriteFile(t, filepath.Join(runWorkDir, "transcribe", "outputs", "transcripts", "raw", "speaker.json"), "{}\n")
mustWriteFile(t, filepath.Join(runWorkDir, "trim", "outputs", "transcripts", "trimmed.json"), "{}\n")
mustWriteFile(t, filepath.Join(runWorkDir, "analyze", "outputs", "artifacts", "session_recap.md"), "# recap\n")
mustWriteFile(t, filepath.Join(runWorkDir, "polish", "reports", "audita.report.json"), "{}\n")
mustWriteFile(t, filepath.Join(runWorkDir, "merge", "config", "seriatim.generated.yml"), "key: value\n")
mustWriteFile(t, filepath.Join(runWorkDir, "logs", "audita.stderr.log"), "stderr\n")
mustWriteFile(t, filepath.Join(runWorkDir, "manifest.json"), "{}\n")
mustWriteFile(t, filepath.Join(sessionRoot, "transcripts", "trimmed.json"), "{}\n")
mustWriteFile(t, filepath.Join(sessionRoot, "artifacts", "session_recap.md"), "# recap\n")
}
type failKeyStore struct {
delegate *storage.FakeBackend
failKey string
}
func (s *failKeyStore) List(ctx context.Context, prefix string) ([]storage.ObjectInfo, error) {
return s.delegate.List(ctx, prefix)
}
func (s *failKeyStore) Download(ctx context.Context, key, localPath string) error {
return s.delegate.Download(ctx, key, localPath)
}
func (s *failKeyStore) Upload(ctx context.Context, localPath, key string, opts storage.UploadOptions) (storage.ObjectInfo, error) {
if strings.TrimSpace(key) == strings.TrimSpace(s.failKey) {
return storage.ObjectInfo{}, errors.New("forced upload failure")
}
return s.delegate.Upload(ctx, localPath, key, opts)
}
func (s *failKeyStore) Exists(ctx context.Context, key string) (bool, error) {
return s.delegate.Exists(ctx, key)
}
func assertExists(t *testing.T, path string) {
t.Helper()
if _, err := os.Stat(path); err != nil {
t.Fatalf("expected path to exist %q: %v", path, err)
}
}
func assertMissing(t *testing.T, path string) {
t.Helper()
if _, err := os.Stat(path); !os.IsNotExist(err) {
t.Fatalf("expected path to be removed %q, stat err=%v", path, err)
}
}

View File

@@ -6,6 +6,7 @@ import (
"fmt"
"io"
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
"gitea.maximumdirect.net/eric/narratio/internal/config"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
@@ -17,10 +18,14 @@ func Resume(ctx context.Context, args []string, out io.Writer) error {
var pipelinePath string
var sessionPath string
var sessionID string
var force bool
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml")
var selectedArtifacts artifactSelectionFlag
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
fs.StringVar(&sessionPath, "session", "", "path to session.yml")
fs.StringVar(&sessionID, "session-id", "", "session identifier for session.yml templates")
fs.BoolVar(&force, "force", false, "force stage execution")
fs.Var(&selectedArtifacts, "artifacts", "artifact names to execute during analyze (comma-separated or repeatable)")
if err := fs.Parse(args); err != nil {
return fmt.Errorf("resume: invalid flags: %w", err)
@@ -28,17 +33,31 @@ func Resume(ctx context.Context, args []string, out io.Writer) error {
if fs.NArg() != 0 {
return fmt.Errorf("resume: unexpected positional arguments")
}
if pipelinePath == "" || sessionPath == "" {
return fmt.Errorf("resume: --config and --session are required")
resolvedPipelinePath, err := resolvePipelineConfigPath(pipelinePath)
if err != nil {
return fmt.Errorf("resume: %w", err)
}
resolvedSessionPath, err := resolveSessionConfigPath(sessionPath)
if err != nil {
return fmt.Errorf("resume: %w", err)
}
cfg, err := config.Load(pipelinePath, sessionPath)
cfg, err := config.LoadWithSessionOptions(resolvedPipelinePath, resolvedSessionPath, config.SessionLoadOptions{
SessionID: sessionID,
})
if err != nil {
return fmt.Errorf("resume: %w", err)
}
if err := config.Validate(cfg); err != nil {
return fmt.Errorf("resume: %w", err)
}
normalizedArtifacts, err := selectedArtifacts.Normalize()
if err != nil {
return fmt.Errorf("resume: invalid --artifacts: %w", err)
}
if err := validateSelectedAnalyzeArtifacts(cfg, normalizedArtifacts); err != nil {
return fmt.Errorf("resume: %w", err)
}
full := BuildFullPlan()
selected := full
@@ -57,7 +76,10 @@ func Resume(ctx context.Context, args []string, out io.Writer) error {
}
}
summary, err := executeStages(ctx, cfg, selected, RunOptions{Force: force})
summary, err := executeStages(ctx, cfg, selected, RunOptions{
Force: force,
SelectedArtifacts: normalizedArtifacts,
})
if err != nil {
return fmt.Errorf("resume: %w", err)
}
@@ -74,7 +96,11 @@ func Resume(ctx context.Context, args []string, out io.Writer) error {
}
func loadManifestIfPresent(ctx context.Context, cfg *config.Config) (*manifest.Manifest, error) {
path := manifestPathFor(cfg)
path := artifacts.SessionManifestPathForCampaign(
cfg.Pipeline.Workspace.Root,
cfg.Session.Campaign,
cfg.Session.SessionID,
)
exists, err := fileExists(path)
if err != nil {
return nil, fmt.Errorf("check manifest %q: %w", path, err)

View File

@@ -16,7 +16,7 @@ import (
func TestResumeStartsAfterCompletedStages(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json")
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{}
m := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
@@ -25,7 +25,7 @@ func TestResumeStartsAfterCompletedStages(t *testing.T) {
if err := store.Save(context.Background(), manifestPath, m); err != nil {
t.Fatalf("save manifest: %v", err)
}
workRoot := filepath.Join(workspaceRoot, "work", "2026-05-03")
workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "raw", "alice.json"), `{"segments":[]}`)
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "speakers.yml"), "match:\n - speaker: Alice\n match: [\"alice\"]\n")
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "autocorrect.yml"), "[]\n")
@@ -52,7 +52,7 @@ func TestResumeStartsAfterCompletedStages(t *testing.T) {
func TestResumeNoRemainingStages(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json")
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{}
m := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
@@ -81,7 +81,7 @@ func TestResumeForceRerunsSucceeded(t *testing.T) {
}))
defer srv.Close()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot, srv.URL)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json")
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
store := &manifest.LocalStore{}
m := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
@@ -105,8 +105,8 @@ func TestResumeForceRerunsSucceeded(t *testing.T) {
func TestRunStageExecutesOnlySelectedStage(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "2026-05-03")
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "merged.json"), `{"segments":[]}`)
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "terms: []\n")
@@ -135,8 +135,8 @@ func TestRunStageExecutesOnlySelectedStage(t *testing.T) {
func TestRunStageSkipAndForce(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "2026-05-03")
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "merged.json"), `{"segments":[]}`)
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "terms: []\n")
@@ -166,11 +166,57 @@ func TestRunStageSkipAndForce(t *testing.T) {
}
}
func TestRunStageForceMarksDownstreamStaleAndResumeContinuesFromStale(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "merged.json"), `{"segments":[]}`)
mustWriteTestFile(t, filepath.Join(workRoot, "inputs", "glossary.yml"), "terms: []\n")
store := &manifest.LocalStore{}
seed := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
for _, name := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "analyze", "archive", "notify"} {
seed.MarkStageSucceeded(name, time.Date(2026, 5, 3, 10, 1, 0, 0, time.UTC), nil)
}
if err := store.Save(context.Background(), manifestPath, seed); err != nil {
t.Fatalf("save manifest: %v", err)
}
var out bytes.Buffer
err := RunStage(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath, "--force", "polish"}, &out)
if err != nil {
t.Fatalf("RunStage(force) error = %v", err)
}
if !strings.Contains(out.String(), "stage=polish executed=1 skipped=0 force=true") {
t.Fatalf("output = %q, want forced polish rerun", out.String())
}
afterForce, err := store.Load(context.Background(), manifestPath)
if err != nil {
t.Fatalf("load manifest after force: %v", err)
}
for _, name := range []string{"normalize", "trim", "analyze", "archive", "notify"} {
if afterForce.Stages[name] == nil || afterForce.Stages[name].Status != manifest.StatusStale {
t.Fatalf("stage %q = %#v, want stale", name, afterForce.Stages[name])
}
}
out.Reset()
err = Resume(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath}, &out)
if err != nil {
t.Fatalf("Resume() error = %v", err)
}
if !strings.Contains(out.String(), "executed=5 skipped=0") {
t.Fatalf("output = %q, want resume to execute normalize..notify", out.String())
}
}
func TestRunStageTrimExecutes(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "2026-05-03")
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "normalized.json"), `{"segments":[{"id":1},{"id":2}]}`)
var out bytes.Buffer
@@ -198,8 +244,8 @@ func TestRunStageTrimExecutes(t *testing.T) {
func TestRunStageNormalizeExecutes(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
manifestPath := filepath.Join(workspaceRoot, "work", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "2026-05-03")
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
workRoot := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03")
mustWriteTestFile(t, filepath.Join(workRoot, "transcripts", "processed.json"), `{"segments":[{"id":1},{"id":2}]}`)
var out bytes.Buffer

View File

@@ -16,10 +16,14 @@ func Run(ctx context.Context, args []string, out io.Writer) error {
var pipelinePath string
var sessionPath string
var sessionID string
var force bool
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml")
var selectedArtifacts artifactSelectionFlag
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
fs.StringVar(&sessionPath, "session", "", "path to session.yml")
fs.StringVar(&sessionID, "session-id", "", "session identifier for session.yml templates")
fs.BoolVar(&force, "force", false, "force stage execution (reserved for future behavior)")
fs.Var(&selectedArtifacts, "artifacts", "artifact names to execute during analyze (comma-separated or repeatable)")
if err := fs.Parse(args); err != nil {
return fmt.Errorf("run: invalid flags: %w", err)
@@ -27,20 +31,37 @@ func Run(ctx context.Context, args []string, out io.Writer) error {
if fs.NArg() != 0 {
return fmt.Errorf("run: unexpected positional arguments")
}
if pipelinePath == "" || sessionPath == "" {
return fmt.Errorf("run: --config and --session are required")
resolvedPipelinePath, err := resolvePipelineConfigPath(pipelinePath)
if err != nil {
return fmt.Errorf("run: %w", err)
}
resolvedSessionPath, err := resolveSessionConfigPath(sessionPath)
if err != nil {
return fmt.Errorf("run: %w", err)
}
cfg, err := config.Load(pipelinePath, sessionPath)
cfg, err := config.LoadWithSessionOptions(resolvedPipelinePath, resolvedSessionPath, config.SessionLoadOptions{
SessionID: sessionID,
})
if err != nil {
return fmt.Errorf("run: %w", err)
}
if err := config.Validate(cfg); err != nil {
return fmt.Errorf("run: %w", err)
}
normalizedArtifacts, err := selectedArtifacts.Normalize()
if err != nil {
return fmt.Errorf("run: invalid --artifacts: %w", err)
}
if err := validateSelectedAnalyzeArtifacts(cfg, normalizedArtifacts); err != nil {
return fmt.Errorf("run: %w", err)
}
stages := BuildFullPlan()
summary, err := executeStages(ctx, cfg, stages, RunOptions{Force: force})
summary, err := executeStages(ctx, cfg, stages, RunOptions{
Force: force,
SelectedArtifacts: normalizedArtifacts,
})
if err != nil {
return fmt.Errorf("run: %w", err)
}

View File

@@ -1,6 +1,8 @@
package app
import (
"time"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
"gitea.maximumdirect.net/eric/narratio/internal/stage"
)
@@ -49,3 +51,43 @@ func firstNonSucceededIndex(stages []stage.Stage, m *manifest.Manifest) int {
}
return len(stages)
}
func canonicalStageNames() []string {
all := stage.All()
out := make([]string, 0, len(all))
for _, s := range all {
if s == nil {
continue
}
out = append(out, s.Name())
}
return out
}
func downstreamStageNames(stageName string) []string {
names := canonicalStageNames()
for i, name := range names {
if name != stageName {
continue
}
return append([]string(nil), names[i+1:]...)
}
return nil
}
func invalidateDownstreamSucceededStages(m *manifest.Manifest, upstreamStage string, at time.Time) []string {
if m == nil || m.Stages == nil {
return nil
}
invalidated := make([]string, 0)
for _, downstream := range downstreamStageNames(upstreamStage) {
sr := m.Stages[downstream]
if sr == nil || sr.Status != manifest.StatusSucceeded {
continue
}
m.MarkStageStale(downstream, at, "upstream stage rerun with force")
invalidated = append(invalidated, downstream)
}
return invalidated
}

View File

@@ -1,6 +1,7 @@
package app
import (
"reflect"
"testing"
"time"
@@ -40,3 +41,48 @@ func TestDecideStageActions(t *testing.T) {
t.Fatalf("forced prepare action = %q, want %q", forced[0].Action, stageActionRun)
}
}
func TestDownstreamStageNames(t *testing.T) {
got := downstreamStageNames("polish")
want := []string{"normalize", "trim", "analyze", "archive", "notify"}
if !reflect.DeepEqual(got, want) {
t.Fatalf("downstreamStageNames(polish) = %#v, want %#v", got, want)
}
missing := downstreamStageNames("unknown")
if len(missing) != 0 {
t.Fatalf("downstreamStageNames(unknown) = %#v, want empty", missing)
}
}
func TestInvalidateDownstreamSucceededStages(t *testing.T) {
now := time.Now().UTC()
m := manifest.New("2026-05-03", now)
m.MarkStageSucceeded("prepare", now, nil)
m.MarkStageSucceeded("transcribe", now, nil)
m.MarkStageSucceeded("merge", now, nil)
m.MarkStageSucceeded("polish", now, nil)
m.MarkStageSucceeded("normalize", now, nil)
m.MarkStageSucceeded("trim", now, nil)
m.MarkStageFailed("analyze", now, "analysis failed")
m.MarkStageSucceeded("archive", now, nil)
m.MarkStageSucceeded("notify", now, nil)
got := invalidateDownstreamSucceededStages(m, "polish", now.Add(1*time.Second))
want := []string{"normalize", "trim", "archive", "notify"}
if !reflect.DeepEqual(got, want) {
t.Fatalf("invalidateDownstreamSucceededStages() = %#v, want %#v", got, want)
}
for _, stageName := range want {
if m.Stages[stageName].Status != manifest.StatusStale {
t.Fatalf("%s status = %q, want stale", stageName, m.Stages[stageName].Status)
}
}
if m.Stages["analyze"].Status != manifest.StatusFailed {
t.Fatalf("analyze status = %q, want failed", m.Stages["analyze"].Status)
}
if m.Stages["prepare"].Status != manifest.StatusSucceeded {
t.Fatalf("prepare status = %q, want succeeded", m.Stages["prepare"].Status)
}
}

View File

@@ -16,10 +16,14 @@ func RunStage(ctx context.Context, args []string, out io.Writer) error {
var pipelinePath string
var sessionPath string
var sessionID string
var force bool
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml")
var selectedArtifacts artifactSelectionFlag
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
fs.StringVar(&sessionPath, "session", "", "path to session.yml")
fs.StringVar(&sessionID, "session-id", "", "session identifier for session.yml templates")
fs.BoolVar(&force, "force", false, "force stage execution (reserved for future behavior)")
fs.Var(&selectedArtifacts, "artifacts", "artifact names to execute during analyze (comma-separated or repeatable)")
if err := fs.Parse(args); err != nil {
return fmt.Errorf("run-stage: invalid flags: %w", err)
@@ -27,25 +31,45 @@ func RunStage(ctx context.Context, args []string, out io.Writer) error {
if fs.NArg() != 1 {
return fmt.Errorf("run-stage: expected exactly one stage name")
}
if pipelinePath == "" || sessionPath == "" {
return fmt.Errorf("run-stage: --config and --session are required")
}
stageName := fs.Arg(0)
normalizedArtifacts, err := selectedArtifacts.Normalize()
if err != nil {
return fmt.Errorf("run-stage: invalid --artifacts: %w", err)
}
if len(normalizedArtifacts) > 0 && stageName != "analyze" {
return fmt.Errorf("run-stage: --artifacts is only supported for stage \"analyze\"")
}
stages, err := BuildSingleStagePlan(stageName)
if err != nil {
return fmt.Errorf("run-stage: %w", err)
}
cfg, err := config.Load(pipelinePath, sessionPath)
resolvedPipelinePath, err := resolvePipelineConfigPath(pipelinePath)
if err != nil {
return fmt.Errorf("run-stage: %w", err)
}
resolvedSessionPath, err := resolveSessionConfigPath(sessionPath)
if err != nil {
return fmt.Errorf("run-stage: %w", err)
}
cfg, err := config.LoadWithSessionOptions(resolvedPipelinePath, resolvedSessionPath, config.SessionLoadOptions{
SessionID: sessionID,
})
if err != nil {
return fmt.Errorf("run-stage: %w", err)
}
if err := config.Validate(cfg); err != nil {
return fmt.Errorf("run-stage: %w", err)
}
if err := validateSelectedAnalyzeArtifacts(cfg, normalizedArtifacts); err != nil {
return fmt.Errorf("run-stage: %w", err)
}
summary, err := executeStages(ctx, cfg, stages, RunOptions{Force: force})
summary, err := executeStages(ctx, cfg, stages, RunOptions{
Force: force,
SelectedArtifacts: normalizedArtifacts,
})
if err != nil {
return fmt.Errorf("run-stage: %w", err)
}

View File

@@ -5,7 +5,6 @@ import (
"fmt"
"log/slog"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/adapters/audita"
@@ -22,16 +21,19 @@ import (
)
type RunOptions struct {
Force bool
Env *Env
Force bool
SelectedArtifacts []string
Env *Env
}
type RunSummary struct {
SessionID string
ManifestPath string
StageNames []string
Executed []string
Skipped []string
SessionID string
RunID string
ManifestPath string
RunManifestPath string
StageNames []string
Executed []string
Skipped []string
}
func executeStages(ctx context.Context, cfg *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
@@ -42,6 +44,7 @@ func executeStages(ctx context.Context, cfg *config.Config, stages []stage.Stage
if env.Config == nil {
env.Config = cfg
}
env.SelectedAnalyzeArtifacts = append([]string(nil), opts.SelectedArtifacts...)
if env.ArtifactStore == nil {
env.ArtifactStore = artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root)
}
@@ -51,6 +54,9 @@ func executeStages(ctx context.Context, cfg *config.Config, stages []stage.Stage
if env.Logger == nil {
env.Logger = logging.NewLogger(os.Stderr, slog.LevelInfo)
}
if _, err := loadSecretsFromConfig(env.Config, env.Logger); err != nil {
return nil, fmt.Errorf("load secrets from files: %w", err)
}
if env.WhisperX == nil {
client, err := buildDefaultWhisperXClient(env.Config)
if err != nil {
@@ -78,17 +84,24 @@ func executeStages(ctx context.Context, cfg *config.Config, stages []stage.Stage
if env.Storage == nil {
env.Storage = &storage.NoopBackend{}
}
if env.ObjectStore == nil && needsObjectStoreForRun(env.Config, stages) {
objectStore, err := storage.NewObjectStoreFromConfig(ctx, env.Config)
if err != nil {
return nil, fmt.Errorf("initialize object store backend: %w", err)
}
env.ObjectStore = objectStore
}
if env.Notifier == nil {
env.Notifier = &notify.NoopSender{}
}
artifactStore := env.ArtifactStore
paths, err := artifactStore.EnsureLayout(cfg.Session.SessionID)
paths, err := artifactStore.EnsureLayoutFor(cfg.Session.Campaign, cfg.Session.SessionID)
if err != nil {
return nil, fmt.Errorf("prepare workdir: %w", err)
}
lock, err := artifactStore.AcquireSessionLock(cfg.Session.SessionID)
lock, err := artifactStore.AcquireSessionLockFor(cfg.Session.Campaign, cfg.Session.SessionID)
if err != nil {
return nil, fmt.Errorf("acquire session lock: %w", err)
}
@@ -101,6 +114,42 @@ func executeStages(ctx context.Context, cfg *config.Config, stages []stage.Stage
if err != nil {
return nil, err
}
runID, err := artifacts.NewRunID()
if err != nil {
return nil, fmt.Errorf("generate run id: %w", err)
}
identityChanged, err := ensureManifestIdentity(cfg, m, runID)
if err != nil {
return nil, fmt.Errorf("initialize manifest identity: %w", err)
}
if identityChanged {
if err := env.ManifestStore.Save(ctx, manifestPath, m); err != nil {
return nil, fmt.Errorf("save manifest identity %q: %w", manifestPath, err)
}
}
runManifestPath := artifacts.SessionRunManifestPathForCampaign(
cfg.Pipeline.Workspace.Root,
cfg.Session.Campaign,
cfg.Session.SessionID,
runID,
)
runManifestStore := &manifest.LocalStore{}
runManifest, err := runManifestStore.CreateRun(
ctx,
cfg.Session.SessionID,
cfg.Session.Campaign,
runID,
opts.Force,
requestedStageNames(stages),
)
if err != nil {
return nil, fmt.Errorf("create run manifest: %w", err)
}
runManifest.SessionManifestPath = manifestPath
syncRunManifestIdentityFromSession(m, runManifest)
if err := runManifestStore.SaveRun(ctx, runManifestPath, runManifest); err != nil {
return nil, fmt.Errorf("save initial run manifest %q: %w", runManifestPath, err)
}
stageEnv := env
@@ -115,12 +164,23 @@ func executeStages(ctx context.Context, cfg *config.Config, stages []stage.Stage
if d.Action == stageActionSkip {
skipped = append(skipped, s.Name())
skipAt := nowUTC()
runManifest.SetStageAction(s.Name(), manifest.RunStageActionSkip, skipAt)
runManifest.MarkStageSkipped(s.Name(), skipAt, "already_succeeded")
if err := runManifestStore.SaveRun(ctx, runManifestPath, runManifest); err != nil {
return nil, fmt.Errorf("save run manifest after skip %q: %w", s.Name(), err)
}
env.Logger.Info("skipping stage", "stage", s.Name(), "reason", "already_succeeded", "force", opts.Force)
continue
}
executed = append(executed, s.Name())
now := nowUTC()
runManifest.SetStageAction(s.Name(), manifest.RunStageActionRun, now)
runManifest.MarkStageRunning(s.Name(), now)
if err := runManifestStore.SaveRun(ctx, runManifestPath, runManifest); err != nil {
return nil, fmt.Errorf("save run manifest before stage %q: %w", s.Name(), err)
}
m.MarkStageRunning(s.Name(), now)
env.Logger.Info("starting stage", "stage", s.Name())
if err := env.ManifestStore.Save(ctx, manifestPath, m); err != nil {
@@ -130,31 +190,65 @@ func executeStages(ctx context.Context, cfg *config.Config, stages []stage.Stage
result, err := s.Run(ctx, stageEnv, m)
if err != nil {
m.MarkStageFailed(s.Name(), nowUTC(), err.Error())
failedAt := nowUTC()
m.MarkStageFailed(s.Name(), failedAt, err.Error())
if saveErr := env.ManifestStore.Save(ctx, manifestPath, m); saveErr != nil {
return nil, fmt.Errorf("stage %q failed (%v) and manifest save failed (%v)", s.Name(), err, saveErr)
}
runManifest.MarkStageFailed(s.Name(), failedAt, err.Error())
syncRunManifestIdentityFromSession(m, runManifest)
if saveErr := runManifestStore.SaveRun(ctx, runManifestPath, runManifest); saveErr != nil {
return nil, fmt.Errorf("stage %q failed (%v) and run-manifest save failed (%v)", s.Name(), err, saveErr)
}
env.Logger.Info("stage failed", "stage", s.Name(), "error", err)
return nil, fmt.Errorf("stage %q failed: %w", s.Name(), err)
}
outputs := mapResultOutputs(result)
m.MarkStageSucceeded(s.Name(), nowUTC(), outputs)
outputs := mapResultOutputs(s.Name(), result, runID)
succeededAt := nowUTC()
m.MarkStageSucceeded(s.Name(), succeededAt, outputs)
applyStageResultToManifest(m, s.Name(), result)
if opts.Force {
invalidateDownstreamSucceededStages(m, s.Name(), succeededAt)
}
if err := env.ManifestStore.Save(ctx, manifestPath, m); err != nil {
return nil, fmt.Errorf("save manifest after stage %q: %w", s.Name(), err)
}
runManifest.MarkStageSucceeded(s.Name(), succeededAt, outputs)
applyStageResultToRunManifest(runManifest, s.Name(), result)
syncRunManifestIdentityFromSession(m, runManifest)
if err := runManifestStore.SaveRun(ctx, runManifestPath, runManifest); err != nil {
return nil, fmt.Errorf("save run manifest after stage %q: %w", s.Name(), err)
}
env.Logger.Debug("manifest saved", "stage", s.Name(), "transition", "succeeded", "path", manifestPath)
env.Logger.Info("stage succeeded", "stage", s.Name())
}
if err := runPostArchiveCleanup(ctx, env, manifestPath, m, executed); err != nil {
failedAt := nowUTC()
runManifest.MarkFailed(failedAt, err.Error())
syncRunManifestIdentityFromSession(m, runManifest)
if saveErr := runManifestStore.SaveRun(ctx, runManifestPath, runManifest); saveErr != nil {
return nil, fmt.Errorf("post-archive cleanup failed (%v) and run-manifest save failed (%v)", err, saveErr)
}
return nil, fmt.Errorf("post-archive cleanup: %w", err)
}
completedAt := nowUTC()
runManifest.MarkSucceeded(completedAt)
syncRunManifestIdentityFromSession(m, runManifest)
if err := runManifestStore.SaveRun(ctx, runManifestPath, runManifest); err != nil {
return nil, fmt.Errorf("save final run manifest %q: %w", runManifestPath, err)
}
return &RunSummary{
SessionID: cfg.Session.SessionID,
ManifestPath: manifestPath,
StageNames: runNames,
Executed: executed,
Skipped: skipped,
SessionID: cfg.Session.SessionID,
RunID: runID,
ManifestPath: manifestPath,
RunManifestPath: runManifestPath,
StageNames: runNames,
Executed: executed,
Skipped: skipped,
}, nil
}
@@ -227,7 +321,7 @@ func buildDefaultAuditaRunner(cfg *config.Config) (audita.Runner, error) {
}
a := cfg.Pipeline.Audita
if strings.TrimSpace(a.Binary) == "" || strings.TrimSpace(a.Timeout) == "" || len(a.Modules) == 0 || strings.TrimSpace(a.BaseURL) == "" || strings.TrimSpace(a.Model) == "" {
if strings.TrimSpace(a.Binary) == "" || strings.TrimSpace(a.Timeout) == "" {
// Compatibility fallback for tests or internal call paths that bypass config validation/defaults.
return &audita.NoopRunner{}, nil
}
@@ -244,7 +338,12 @@ func buildDefaultAuditaRunner(cfg *config.Config) (audita.Runner, error) {
append([]string(nil), a.Modules...),
a.BaseURL,
a.Model,
a.LLMConcurrency,
a.TranscriptDescription,
a.ConfigPath,
a.OutputSchema,
a.WorkDirRetention,
a.TotalLLMConcurrency,
a.ProposalLLMConcurrency,
a.ValidationModel,
a.ValidationLLMConcurrency,
report,
@@ -289,22 +388,31 @@ func fileExists(path string) (bool, error) {
return false, err
}
func mapResultOutputs(result *stage.StageResult) []manifest.ArtifactRecord {
func mapResultOutputs(stageName string, result *stage.StageResult, runID string) []manifest.ArtifactRecord {
if result == nil || len(result.Outputs) == 0 {
return nil
}
runID = strings.TrimSpace(runID)
out := make([]manifest.ArtifactRecord, 0, len(result.Outputs))
for _, ref := range result.Outputs {
localPath := ref.AbsolutePath
if localPath == "" {
localPath = ref.RelativePath
}
kind := ref.Kind
sourceID := ""
if stageName == "analyze" {
sourceID = artifacts.ConfiguredArtifactSourceID(ref.Kind)
kind = "scriptorium_artifact"
}
out = append(out, manifest.ArtifactRecord{
Kind: ref.Kind,
LocalPath: localPath,
RemoteKey: ref.RemoteKey,
Checksum: ref.Checksum,
Kind: kind,
SourceID: sourceID,
LocalPath: localPath,
ProducerRunID: runID,
RemoteKey: ref.RemoteKey,
Checksum: ref.Checksum,
})
}
@@ -330,6 +438,132 @@ func applyStageResultToManifest(m *manifest.Manifest, stageName string, result *
}
}
func manifestPathFor(cfg *config.Config) string {
return filepath.Join(cfg.Pipeline.Workspace.Root, "work", cfg.Session.SessionID, "manifest.json")
func ensureManifestIdentity(cfg *config.Config, m *manifest.Manifest, runID string) (bool, error) {
if cfg == nil || cfg.Pipeline == nil || cfg.Session == nil || m == nil {
return false, nil
}
changed := false
campaign := strings.TrimSpace(cfg.Session.Campaign)
sessionID := strings.TrimSpace(cfg.Session.SessionID)
if sessionID == "" {
sessionID = strings.TrimSpace(m.SessionID)
}
if m.Campaign == "" && campaign != "" {
m.Campaign = campaign
changed = true
}
runID = strings.TrimSpace(runID)
if runID != "" && m.RunID != runID {
m.RunID = runID
changed = true
}
if m.LocalWorkDir == "" && campaign != "" && sessionID != "" && m.RunID != "" {
m.LocalWorkDir = artifacts.SessionRunRootForCampaign(cfg.Pipeline.Workspace.Root, campaign, sessionID, m.RunID)
changed = true
}
if m.LocalSpoolDir == "" && campaign != "" && sessionID != "" && m.RunID != "" && strings.TrimSpace(cfg.Pipeline.Spool.Root) != "" {
m.LocalSpoolDir = artifacts.SessionSpoolAudioDir(cfg.Pipeline.Spool.Root, campaign, sessionID, m.RunID)
changed = true
}
if cfg.Pipeline.Storage.S3 != nil {
if m.S3Bucket == "" && strings.TrimSpace(cfg.Pipeline.Storage.S3.Bucket) != "" {
m.S3Bucket = strings.TrimSpace(cfg.Pipeline.Storage.S3.Bucket)
changed = true
}
sessionPrefix := artifacts.S3SessionPrefix(cfg.Pipeline.Storage.S3.RootPrefix, campaign, sessionID)
if m.S3SessionPrefix == "" && sessionPrefix != "" {
m.S3SessionPrefix = sessionPrefix
changed = true
}
runPrefix := artifacts.S3RunPrefix(sessionPrefix, m.RunID)
if m.S3RunPrefix == "" && runPrefix != "" {
m.S3RunPrefix = runPrefix
changed = true
}
}
return changed, nil
}
func requestedStageNames(stages []stage.Stage) []string {
out := make([]string, 0, len(stages))
for _, s := range stages {
if s == nil {
continue
}
out = append(out, s.Name())
}
return out
}
func applyStageResultToRunManifest(m *manifest.RunManifest, stageName string, result *stage.StageResult) {
if m == nil || result == nil {
return
}
sr := m.Stages[stageName]
if sr == nil {
return
}
if len(result.Logs) > 0 {
sr.Logs = append([]string(nil), result.Logs...)
}
if len(result.GeneratedConfigs) > 0 {
sr.GeneratedConfigs = append([]string(nil), result.GeneratedConfigs...)
}
if len(result.Metadata) > 0 {
sr.Metadata = result.Metadata
}
}
func syncRunManifestIdentityFromSession(session *manifest.Manifest, run *manifest.RunManifest) {
if session == nil || run == nil {
return
}
run.Campaign = session.Campaign
run.LocalWorkDir = session.LocalWorkDir
run.LocalSpoolDir = session.LocalSpoolDir
run.S3Bucket = session.S3Bucket
run.S3SessionPrefix = session.S3SessionPrefix
run.S3RunPrefix = session.S3RunPrefix
}
func manifestPathFor(cfg *config.Config) string {
return artifacts.SessionManifestPathForCampaign(
cfg.Pipeline.Workspace.Root,
cfg.Session.Campaign,
cfg.Session.SessionID,
)
}
func needsObjectStoreForRun(cfg *config.Config, stages []stage.Stage) bool {
if cfg == nil || cfg.Pipeline == nil || cfg.Session == nil {
return false
}
stageRequested := func(name string) bool {
for _, s := range stages {
if s != nil && s.Name() == name {
return true
}
}
return false
}
if cfg.Session.Inputs.AudioS3 != nil && stageRequested("prepare") {
return true
}
if !stageRequested("archive") {
return false
}
if cfg.Pipeline.Archive == nil {
return false
}
if cfg.Pipeline.Archive.Enabled != nil && !*cfg.Pipeline.Archive.Enabled {
return false
}
if cfg.Pipeline.Archive.UploadRun != nil && !*cfg.Pipeline.Archive.UploadRun {
return false
}
return true
}

View File

@@ -3,6 +3,7 @@ package app
import (
"context"
"errors"
"fmt"
"os"
"path/filepath"
"strings"
@@ -44,6 +45,213 @@ func (s countingStage) Run(_ context.Context, _ *stage.Env, _ *manifest.Manifest
return &stage.StageResult{Metadata: map[string]any{"counting": true}}, nil
}
type captureSelectedArtifactsStage struct {
name string
captured *[]string
}
func (s captureSelectedArtifactsStage) Name() string { return s.name }
func (s captureSelectedArtifactsStage) Declares() stage.IODecl { return stage.IODecl{} }
func (s captureSelectedArtifactsStage) Run(_ context.Context, env *stage.Env, _ *manifest.Manifest) (*stage.StageResult, error) {
if s.captured != nil {
*s.captured = append((*s.captured)[:0], env.SelectedAnalyzeArtifacts...)
}
return &stage.StageResult{Metadata: map[string]any{"captured": true}}, nil
}
type analyzeOutputStage struct {
output artifacts.Ref
}
func (s analyzeOutputStage) Name() string { return "analyze" }
func (s analyzeOutputStage) Declares() stage.IODecl { return stage.IODecl{} }
func (s analyzeOutputStage) Run(_ context.Context, _ *stage.Env, _ *manifest.Manifest) (*stage.StageResult, error) {
return &stage.StageResult{
Outputs: []artifacts.Ref{s.output},
}, nil
}
type selectedAnalyzeArtifactStage struct {
expected []string
}
func (s selectedAnalyzeArtifactStage) Name() string { return "analyze" }
func (s selectedAnalyzeArtifactStage) Declares() stage.IODecl { return stage.IODecl{} }
func (s selectedAnalyzeArtifactStage) Run(_ context.Context, env *stage.Env, m *manifest.Manifest) (*stage.StageResult, error) {
if len(env.SelectedAnalyzeArtifacts) != len(s.expected) {
return nil, fmt.Errorf("selected artifacts len = %d, want %d", len(env.SelectedAnalyzeArtifacts), len(s.expected))
}
for i := range s.expected {
if env.SelectedAnalyzeArtifacts[i] != s.expected[i] {
return nil, fmt.Errorf("selected artifacts[%d] = %q, want %q", i, env.SelectedAnalyzeArtifacts[i], s.expected[i])
}
}
outputPath := filepath.Join(
artifacts.SessionWorkDirForCampaign(env.Config.Pipeline.Workspace.Root, env.Config.Session.Campaign, m.SessionID),
"artifacts",
"player_handout.md",
)
if err := os.MkdirAll(filepath.Dir(outputPath), 0o755); err != nil {
return nil, fmt.Errorf("mkdir artifact dir: %w", err)
}
if err := os.WriteFile(outputPath, []byte("player handout\n"), 0o644); err != nil {
return nil, fmt.Errorf("write player handout: %w", err)
}
return &stage.StageResult{
Outputs: []artifacts.Ref{
{
Kind: "player_handout",
Category: "artifacts",
RelativePath: "artifacts/player_handout.md",
AbsolutePath: outputPath,
},
},
Metadata: map[string]any{
"stage": "analyze",
},
}, nil
}
func TestExecuteStagesPropagatesSelectedArtifactsToEnv(t *testing.T) {
cfg := testConfig(t)
captured := []string{}
stageToRun := captureSelectedArtifactsStage{name: "analyze", captured: &captured}
_, err := executeStages(context.Background(), cfg, []stage.Stage{stageToRun}, RunOptions{
SelectedArtifacts: []string{"player_handout", "session_recap"},
})
if err != nil {
t.Fatalf("executeStages() error = %v", err)
}
if len(captured) != 2 {
t.Fatalf("captured len = %d, want 2 (%v)", len(captured), captured)
}
if captured[0] != "player_handout" || captured[1] != "session_recap" {
t.Fatalf("captured = %v, want [player_handout session_recap]", captured)
}
}
func TestExecuteStagesAnalyzeOutputsPersistAsScriptoriumArtifacts(t *testing.T) {
cfg := testConfig(t)
storeForPaths := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root)
sessionPaths := storeForPaths.SessionPathsFor(cfg.Session.Campaign, cfg.Session.SessionID)
outputPath := filepath.Join(sessionPaths.ArtifactsDir, "session_recap.md")
stageToRun := analyzeOutputStage{
output: artifacts.Ref{
Kind: "session_recap",
Category: "artifacts",
RelativePath: "artifacts/session_recap.md",
AbsolutePath: outputPath,
},
}
summary, err := executeStages(context.Background(), cfg, []stage.Stage{stageToRun}, RunOptions{})
if err != nil {
t.Fatalf("executeStages() error = %v", err)
}
store := &manifest.LocalStore{}
sessionManifest, err := store.Load(context.Background(), summary.ManifestPath)
if err != nil {
t.Fatalf("load session manifest: %v", err)
}
sessionStage := sessionManifest.Stages["analyze"]
if sessionStage == nil {
t.Fatal("session manifest analyze stage missing")
}
if len(sessionStage.Outputs) != 1 {
t.Fatalf("session analyze outputs len = %d, want 1", len(sessionStage.Outputs))
}
sessionOutput := sessionStage.Outputs[0]
if sessionOutput.Kind != "scriptorium_artifact" {
t.Fatalf("session output kind = %q, want scriptorium_artifact", sessionOutput.Kind)
}
if sessionOutput.SourceID != "narratio.artifact.session_recap" {
t.Fatalf("session output source_id = %q, want narratio.artifact.session_recap", sessionOutput.SourceID)
}
if sessionOutput.LocalPath != outputPath {
t.Fatalf("session output local_path = %q, want %q", sessionOutput.LocalPath, outputPath)
}
runManifest, err := store.LoadRun(context.Background(), summary.RunManifestPath)
if err != nil {
t.Fatalf("load run manifest: %v", err)
}
runStage := runManifest.Stages["analyze"]
if runStage == nil {
t.Fatal("run manifest analyze stage missing")
}
if len(runStage.Outputs) != 1 {
t.Fatalf("run analyze outputs len = %d, want 1", len(runStage.Outputs))
}
runOutput := runStage.Outputs[0]
if runOutput.Kind != "scriptorium_artifact" {
t.Fatalf("run output kind = %q, want scriptorium_artifact", runOutput.Kind)
}
if runOutput.SourceID != "narratio.artifact.session_recap" {
t.Fatalf("run output source_id = %q, want narratio.artifact.session_recap", runOutput.SourceID)
}
if runOutput.LocalPath != outputPath {
t.Fatalf("run output local_path = %q, want %q", runOutput.LocalPath, outputPath)
}
}
func TestExecuteStagesArchiveFailsWhenRequiredRecapPromotionMissingForSelectedArtifacts(t *testing.T) {
cfg := testConfig(t)
cfg.Pipeline.Storage.S3 = &config.StorageS3Config{
Bucket: "my-dnd-archive",
RootPrefix: "dnd",
}
cfg.Pipeline.Archive = &config.ArchiveConfig{
Enabled: boolPtr(true),
UploadRun: boolPtr(true),
PromoteArtifacts: []config.ArchivePromotionRule{
{From: "artifacts/session_recap.md", To: "artifacts/session_recap.md", Required: boolPtr(true)},
},
}
store := &manifest.LocalStore{}
manifestPath := manifestPathFor(cfg)
seed := manifest.New(cfg.Session.SessionID, time.Now().UTC())
seed.Campaign = cfg.Session.Campaign
for _, stageName := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim"} {
seed.MarkStageSucceeded(stageName, time.Now().UTC(), nil)
}
if err := os.MkdirAll(filepath.Dir(manifestPath), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
if err := store.Save(context.Background(), manifestPath, seed); err != nil {
t.Fatalf("Save manifest error = %v", err)
}
archiveStageImpl, err := stage.Select("archive")
if err != nil {
t.Fatalf("Select(archive) error = %v", err)
}
_, err = executeStages(
context.Background(),
cfg,
[]stage.Stage{
selectedAnalyzeArtifactStage{expected: []string{"player_handout"}},
archiveStageImpl,
},
RunOptions{
SelectedArtifacts: []string{"player_handout"},
Env: &Env{ObjectStore: &storage.FakeBackend{}},
},
)
if err == nil {
t.Fatal("expected archive promotion failure, got nil")
}
if !strings.Contains(err.Error(), "required promotion source missing") {
t.Fatalf("error = %q, want required promotion source missing", err.Error())
}
}
func TestExecuteStagesPlaceholderSuccessUpdatesManifest(t *testing.T) {
cfg := testConfig(t)
@@ -144,6 +352,15 @@ func TestExecuteStagesPlaceholderSuccessUpdatesManifest(t *testing.T) {
}
continue
}
if name == "archive" {
if sr.Metadata == nil || sr.Metadata["stage"] != "archive" {
t.Fatalf("archive metadata missing stage=archive: %#v", sr.Metadata)
}
if sr.Metadata["skipped"] != true {
t.Fatalf("archive metadata missing skipped=true for test config without archive section: %#v", sr.Metadata)
}
continue
}
if sr.Metadata == nil || sr.Metadata["placeholder"] != true {
t.Fatalf("stage %q missing placeholder metadata", name)
}
@@ -226,6 +443,56 @@ func TestExecuteStagesForceRerunsSucceeded(t *testing.T) {
}
}
func TestExecuteStagesForceSuccessInvalidatesDownstreamSucceededStages(t *testing.T) {
cfg := testConfig(t)
manifestPath := manifestPathFor(cfg)
store := &manifest.LocalStore{}
existing := manifest.New(cfg.Session.SessionID, time.Date(2026, 5, 3, 1, 0, 0, 0, time.UTC))
for _, stageName := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "archive", "notify"} {
existing.MarkStageSucceeded(stageName, time.Date(2026, 5, 3, 1, 1, 0, 0, time.UTC), nil)
}
existing.MarkStageFailed("analyze", time.Date(2026, 5, 3, 1, 1, 0, 0, time.UTC), "previous analyze failure")
if err := os.MkdirAll(filepath.Dir(manifestPath), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
if err := store.Save(context.Background(), manifestPath, existing); err != nil {
t.Fatalf("Save manifest error = %v", err)
}
runs := 0
stageToRun := countingStage{name: "polish", runs: &runs}
summary, err := executeStages(context.Background(), cfg, []stage.Stage{stageToRun}, RunOptions{Force: true})
if err != nil {
t.Fatalf("executeStages() error = %v", err)
}
if runs != 1 {
t.Fatalf("runs = %d, want 1 with force", runs)
}
if len(summary.Executed) != 1 || summary.Executed[0] != "polish" || len(summary.Skipped) != 0 {
t.Fatalf("summary = %#v, want executed polish", summary)
}
loaded, err := store.Load(context.Background(), manifestPath)
if err != nil {
t.Fatalf("Load manifest error = %v", err)
}
if loaded.Stages["polish"] == nil || loaded.Stages["polish"].Status != manifest.StatusSucceeded {
t.Fatalf("polish status = %#v, want succeeded", loaded.Stages["polish"])
}
for _, stageName := range []string{"normalize", "trim", "archive", "notify"} {
if loaded.Stages[stageName] == nil || loaded.Stages[stageName].Status != manifest.StatusStale {
t.Fatalf("%s status = %#v, want stale", stageName, loaded.Stages[stageName])
}
}
if loaded.Stages["analyze"] == nil || loaded.Stages["analyze"].Status != manifest.StatusFailed {
t.Fatalf("analyze status = %#v, want preserved failed", loaded.Stages["analyze"])
}
if loaded.Stages["transcribe"] == nil || loaded.Stages["transcribe"].Status != manifest.StatusSucceeded {
t.Fatalf("transcribe status = %#v, want preserved succeeded", loaded.Stages["transcribe"])
}
}
func TestExecuteStagesFailureUpdatesManifest(t *testing.T) {
cfg := testConfig(t)
@@ -271,7 +538,14 @@ func TestExecuteStagesLoadsExistingManifest(t *testing.T) {
existing := manifest.New(cfg.Session.SessionID, time.Date(2026, 5, 3, 1, 0, 0, 0, time.UTC))
existing.MarkStageSucceeded("prepare", time.Date(2026, 5, 3, 1, 1, 0, 0, time.UTC), nil)
audioPath := filepath.Join(cfg.Pipeline.Workspace.Root, "work", cfg.Session.SessionID, "audio", "alice.flac")
audioPath := filepath.Join(
cfg.Pipeline.Workspace.Root,
"work",
cfg.Session.Campaign,
cfg.Session.SessionID,
"audio",
"alice.flac",
)
if err := os.MkdirAll(filepath.Dir(audioPath), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
@@ -306,6 +580,208 @@ func TestExecuteStagesLoadsExistingManifest(t *testing.T) {
}
}
func TestExecuteStagesCreatesRunManifestPerInvocation(t *testing.T) {
cfg := testConfig(t)
run1, err := executeStages(context.Background(), cfg, []stage.Stage{BuildFullPlan()[0]}, RunOptions{})
if err != nil {
t.Fatalf("first executeStages() error = %v", err)
}
run2, err := executeStages(context.Background(), cfg, []stage.Stage{BuildFullPlan()[0]}, RunOptions{Force: true})
if err != nil {
t.Fatalf("second executeStages() error = %v", err)
}
if run1.RunID == "" || run2.RunID == "" {
t.Fatalf("run ids must be set, got %q and %q", run1.RunID, run2.RunID)
}
if run1.RunID == run2.RunID {
t.Fatalf("expected distinct run ids, got %q", run1.RunID)
}
if run1.RunManifestPath == "" || run2.RunManifestPath == "" {
t.Fatalf("run manifest paths must be set, got %q and %q", run1.RunManifestPath, run2.RunManifestPath)
}
if run1.RunManifestPath == run2.RunManifestPath {
t.Fatalf("expected distinct run manifest paths, got %q", run1.RunManifestPath)
}
for _, path := range []string{run1.RunManifestPath, run2.RunManifestPath} {
if _, statErr := os.Stat(path); statErr != nil {
t.Fatalf("run manifest missing at %q: %v", path, statErr)
}
}
store := &manifest.LocalStore{}
sessionManifest, err := store.Load(context.Background(), run2.ManifestPath)
if err != nil {
t.Fatalf("Load session manifest error = %v", err)
}
if sessionManifest.RunID != run2.RunID {
t.Fatalf("session manifest run_id = %q, want latest run id %q", sessionManifest.RunID, run2.RunID)
}
}
func TestExecuteStagesRunManifestRecordsSkippedStage(t *testing.T) {
cfg := testConfig(t)
store := &manifest.LocalStore{}
manifestPath := manifestPathFor(cfg)
existing := manifest.New(cfg.Session.SessionID, time.Date(2026, 5, 3, 1, 0, 0, 0, time.UTC))
existing.MarkStageSucceeded("transcribe", time.Date(2026, 5, 3, 1, 1, 0, 0, time.UTC), nil)
if err := os.MkdirAll(filepath.Dir(manifestPath), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
if err := store.Save(context.Background(), manifestPath, existing); err != nil {
t.Fatalf("Save manifest error = %v", err)
}
summary, err := executeStages(context.Background(), cfg, []stage.Stage{BuildFullPlan()[1]}, RunOptions{})
if err != nil {
t.Fatalf("executeStages() error = %v", err)
}
if len(summary.Skipped) != 1 || summary.Skipped[0] != "transcribe" {
t.Fatalf("summary = %#v, want skipped transcribe", summary)
}
runManifest, err := store.LoadRun(context.Background(), summary.RunManifestPath)
if err != nil {
t.Fatalf("LoadRun() error = %v", err)
}
sr := runManifest.Stages["transcribe"]
if sr == nil {
t.Fatal("run manifest transcribe stage missing")
}
if sr.Action != manifest.RunStageActionSkip {
t.Fatalf("action = %q, want %q", sr.Action, manifest.RunStageActionSkip)
}
if sr.Status != manifest.StatusSkipped {
t.Fatalf("status = %q, want %q", sr.Status, manifest.StatusSkipped)
}
}
func TestExecuteStagesRunLocalArtifactsAndCanonicalPromotion(t *testing.T) {
cfg := testConfig(t)
stages := []stage.Stage{
BuildFullPlan()[0], // prepare
BuildFullPlan()[1], // transcribe
BuildFullPlan()[2], // merge
BuildFullPlan()[3], // polish
BuildFullPlan()[4], // normalize
BuildFullPlan()[5], // trim
}
summary, err := executeStages(context.Background(), cfg, stages, RunOptions{})
if err != nil {
t.Fatalf("executeStages() error = %v", err)
}
if summary.RunID == "" {
t.Fatal("run id must be set")
}
runRoot := artifacts.SessionRunRootForCampaign(
cfg.Pipeline.Workspace.Root,
cfg.Session.Campaign,
cfg.Session.SessionID,
summary.RunID,
)
paths := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root).SessionPathsFor(cfg.Session.Campaign, cfg.Session.SessionID)
runLocalChecks := []string{
filepath.Join(runRoot, "transcribe", "outputs", "transcripts", "raw", "alice.json"),
filepath.Join(runRoot, "merge", "logs", "seriatim.stdout.log"),
filepath.Join(runRoot, "polish", "config", "audita.generated.yml"),
filepath.Join(runRoot, "normalize", "logs", "seriatim.normalize.stdout.log"),
filepath.Join(runRoot, "trim", "outputs", "transcripts", "trimmed.json"),
}
for _, p := range runLocalChecks {
if _, statErr := os.Stat(p); statErr != nil {
t.Fatalf("run-local artifact missing at %q: %v", p, statErr)
}
}
canonicalChecks := []string{
filepath.Join(paths.TranscriptsRawDir, "alice.json"),
filepath.Join(paths.TranscriptsDir, "merged.json"),
filepath.Join(paths.TranscriptsDir, "processed.json"),
filepath.Join(paths.TranscriptsDir, "normalized.json"),
filepath.Join(paths.TranscriptsDir, "trimmed.json"),
}
for _, p := range canonicalChecks {
if _, statErr := os.Stat(p); statErr != nil {
t.Fatalf("canonical promoted artifact missing at %q: %v", p, statErr)
}
}
store := &manifest.LocalStore{}
sessionManifest, err := store.Load(context.Background(), summary.ManifestPath)
if err != nil {
t.Fatalf("Load manifest error = %v", err)
}
if got := sessionManifest.Stages["trim"]; got == nil || len(got.Outputs) == 0 {
t.Fatalf("trim stage outputs missing in session manifest: %#v", got)
}
for _, out := range sessionManifest.Stages["trim"].Outputs {
if strings.Contains(out.LocalPath, string(filepath.Separator)+"runs"+string(filepath.Separator)) {
t.Fatalf("session manifest output should be canonical, got run-local path %q", out.LocalPath)
}
if out.ProducerRunID != summary.RunID {
t.Fatalf("producer_run_id = %q, want %q", out.ProducerRunID, summary.RunID)
}
}
runManifest, err := store.LoadRun(context.Background(), summary.RunManifestPath)
if err != nil {
t.Fatalf("LoadRun() error = %v", err)
}
mergeStage := runManifest.Stages["merge"]
if mergeStage == nil || len(mergeStage.Logs) == 0 {
t.Fatalf("merge logs missing in run manifest: %#v", mergeStage)
}
for _, logPath := range mergeStage.Logs {
if !strings.Contains(logPath, filepath.Join("runs", summary.RunID, "merge", "logs")) {
t.Fatalf("run manifest merge log path = %q, want run-local merge logs path", logPath)
}
}
}
func TestExecuteStagesSkippedStagePreservesExistingOutputsProvenance(t *testing.T) {
cfg := testConfig(t)
store := &manifest.LocalStore{}
manifestPath := manifestPathFor(cfg)
existing := manifest.New(cfg.Session.SessionID, time.Date(2026, 5, 3, 1, 0, 0, 0, time.UTC))
existing.MarkStageSucceeded("transcribe", time.Date(2026, 5, 3, 1, 1, 0, 0, time.UTC), []manifest.ArtifactRecord{
{
Kind: "transcript_raw",
LocalPath: "transcripts/raw/alice.json",
ProducerRunID: "20260501T000000Z-deadbeef",
},
})
if err := os.MkdirAll(filepath.Dir(manifestPath), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
if err := store.Save(context.Background(), manifestPath, existing); err != nil {
t.Fatalf("Save manifest error = %v", err)
}
summary, err := executeStages(context.Background(), cfg, []stage.Stage{BuildFullPlan()[1]}, RunOptions{})
if err != nil {
t.Fatalf("executeStages() error = %v", err)
}
if len(summary.Skipped) != 1 || summary.Skipped[0] != "transcribe" {
t.Fatalf("summary = %#v, want skipped transcribe", summary)
}
loaded, err := store.Load(context.Background(), manifestPath)
if err != nil {
t.Fatalf("Load manifest error = %v", err)
}
got := loaded.Stages["transcribe"]
if got == nil || len(got.Outputs) != 1 {
t.Fatalf("transcribe outputs = %#v, want one preserved output", got)
}
if got.Outputs[0].ProducerRunID != "20260501T000000Z-deadbeef" {
t.Fatalf("producer_run_id = %q, want preserved value", got.Outputs[0].ProducerRunID)
}
}
func TestAdapterBackedStageFailureMarksManifestFailed(t *testing.T) {
cases := []struct {
name string
@@ -315,7 +791,7 @@ func TestAdapterBackedStageFailureMarksManifestFailed(t *testing.T) {
{name: "merge", env: &Env{Seriatim: &seriatim.FakeRunner{Err: errors.New("merge fail")}}},
{name: "polish", env: &Env{Audita: &audita.FakeRunner{Err: errors.New("polish fail")}}},
{name: "analyze", env: &Env{Scriptorium: &scriptorium.FakeRunner{RunErr: errors.New("analyze fail")}}},
{name: "archive", env: &Env{Storage: &storage.FakeBackend{Err: errors.New("archive fail")}}},
{name: "archive", env: &Env{ObjectStore: &storage.FakeBackend{UploadErr: errors.New("archive fail")}}},
{name: "notify", env: &Env{Notifier: &notify.FakeSender{Err: errors.New("notify fail")}}},
}
@@ -332,7 +808,7 @@ func TestAdapterBackedStageFailureMarksManifestFailed(t *testing.T) {
tc.env.ArtifactStore = artifactStore
tc.env.ManifestStore = &manifest.LocalStore{}
if tc.name == "transcribe" {
paths, ensureErr := artifactStore.EnsureLayout(cfg.Session.SessionID)
paths, ensureErr := artifactStore.EnsureLayoutFor(cfg.Session.Campaign, cfg.Session.SessionID)
if ensureErr != nil {
t.Fatalf("EnsureLayout() error = %v", ensureErr)
}
@@ -347,7 +823,7 @@ func TestAdapterBackedStageFailureMarksManifestFailed(t *testing.T) {
}
}
if tc.name == "merge" {
paths, ensureErr := artifactStore.EnsureLayout(cfg.Session.SessionID)
paths, ensureErr := artifactStore.EnsureLayoutFor(cfg.Session.Campaign, cfg.Session.SessionID)
if ensureErr != nil {
t.Fatalf("EnsureLayout() error = %v", ensureErr)
}
@@ -366,7 +842,7 @@ func TestAdapterBackedStageFailureMarksManifestFailed(t *testing.T) {
}
}
if tc.name == "polish" {
paths, ensureErr := artifactStore.EnsureLayout(cfg.Session.SessionID)
paths, ensureErr := artifactStore.EnsureLayoutFor(cfg.Session.Campaign, cfg.Session.SessionID)
if ensureErr != nil {
t.Fatalf("EnsureLayout() error = %v", ensureErr)
}
@@ -378,7 +854,7 @@ func TestAdapterBackedStageFailureMarksManifestFailed(t *testing.T) {
}
}
if tc.name == "analyze" {
paths, ensureErr := artifactStore.EnsureLayout(cfg.Session.SessionID)
paths, ensureErr := artifactStore.EnsureLayoutFor(cfg.Session.Campaign, cfg.Session.SessionID)
if ensureErr != nil {
t.Fatalf("EnsureLayout() error = %v", ensureErr)
}
@@ -394,12 +870,47 @@ func TestAdapterBackedStageFailureMarksManifestFailed(t *testing.T) {
PromptID: "dnd.session_recap",
OutputPath: "artifacts/session_recap.md",
Inputs: map[string]config.ScriptoriumInputConfig{
"transcript": {Source: "processed_transcript", Required: true},
"transcript": {Source: "narratio.transcript.polished", Required: true},
},
},
},
}
}
if tc.name == "archive" {
cfg.Pipeline.Archive = &config.ArchiveConfig{
Enabled: boolPtr(true),
UploadRun: boolPtr(true),
}
cfg.Pipeline.Storage.S3 = &config.StorageS3Config{
Bucket: "my-dnd-archive",
RootPrefix: "dnd",
}
runID := "20260516T010203Z-0a1b2c3d"
runWorkDir := filepath.Join(cfg.Pipeline.Workspace.Root, "work", cfg.Session.Campaign, cfg.Session.SessionID, runID)
if err := os.MkdirAll(filepath.Join(runWorkDir, "inputs"), 0o755); err != nil {
t.Fatalf("mkdir archive inputs dir: %v", err)
}
if err := os.WriteFile(filepath.Join(runWorkDir, "inputs", "session.yml"), []byte("session_id: 2026-05-03\n"), 0o644); err != nil {
t.Fatalf("write archive fixture session.yml: %v", err)
}
if err := os.WriteFile(filepath.Join(runWorkDir, "manifest.json"), []byte("{}\n"), 0o644); err != nil {
t.Fatalf("write archive fixture manifest.json: %v", err)
}
seed := manifest.New(cfg.Session.SessionID, time.Now().UTC())
seed.Campaign = cfg.Session.Campaign
seed.RunID = runID
seed.LocalWorkDir = runWorkDir
seed.S3Bucket = "my-dnd-archive"
seed.S3SessionPrefix = "dnd/campaigns/" + cfg.Session.Campaign + "/sessions/" + cfg.Session.SessionID + "/"
seed.S3RunPrefix = seed.S3SessionPrefix + "runs/" + runID + "/"
for _, name := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "analyze"} {
seed.MarkStageSucceeded(name, time.Now().UTC(), nil)
}
if err := tc.env.ManifestStore.Save(context.Background(), manifestPathFor(cfg), seed); err != nil {
t.Fatalf("seed archive manifest: %v", err)
}
}
_, runErr := executeStages(context.Background(), cfg, []stage.Stage{selected}, RunOptions{Env: tc.env})
if runErr == nil {
@@ -430,7 +941,7 @@ func testConfig(t *testing.T) *config.Config {
pipelinePath := filepath.Join(cfgDir, "pipeline.yml")
mustWriteFile(t, pipelinePath, "workspace:\n root: "+workspace+"\n")
mustWriteFile(t, sessionPath, "session_id: 2026-05-03\n")
mustWriteFile(t, sessionPath, "session_id: 2026-05-03\ncampaign: sample-campaign\n")
mustWriteFile(t, filepath.Join(cfgDir, "speakers.yml"), "alice: alice.flac\n")
mustWriteFile(t, filepath.Join(cfgDir, "autocorrect.yml"), "[]\n")
mustWriteFile(t, filepath.Join(cfgDir, "glossary.yml"), "[]\n")
@@ -442,6 +953,7 @@ func testConfig(t *testing.T) *config.Config {
SessionPath: sessionPath,
Session: &config.SessionConfig{
SessionID: "2026-05-03",
Campaign: "sample-campaign",
Inputs: config.SessionInputsConfig{
AudioDir: "./audio",
SpeakersFile: "./speakers.yml",
@@ -452,6 +964,55 @@ func testConfig(t *testing.T) *config.Config {
}
}
func TestBuildDefaultRunnersWithOmittedToolSections(t *testing.T) {
dir := t.TempDir()
pipelinePath := filepath.Join(dir, "pipeline.yml")
sessionPath := filepath.Join(dir, "session.yml")
pipelineYAML := `workspace:
root: ` + t.TempDir() + `
whisperx:
transcribe_url: https://example.com/transcribe
analyzer:
timeout: 20m
notification:
timeout: 10s
`
sessionYAML := `session_id: 2026-05-03
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
mustWriteFile(t, pipelinePath, pipelineYAML)
mustWriteFile(t, sessionPath, sessionYAML)
cfg, err := config.Load(pipelinePath, sessionPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
if err := config.Validate(cfg); err != nil {
t.Fatalf("Validate() error = %v", err)
}
serRunner, err := buildDefaultSeriatimRunner(cfg)
if err != nil {
t.Fatalf("buildDefaultSeriatimRunner() error = %v", err)
}
if _, ok := serRunner.(*seriatim.SubprocessRunner); !ok {
t.Fatalf("seriatim runner type = %T, want *seriatim.SubprocessRunner", serRunner)
}
audRunner, err := buildDefaultAuditaRunner(cfg)
if err != nil {
t.Fatalf("buildDefaultAuditaRunner() error = %v", err)
}
if _, ok := audRunner.(*audita.SubprocessRunner); !ok {
t.Fatalf("audita runner type = %T, want *audita.SubprocessRunner", audRunner)
}
}
func mustWriteFile(t *testing.T, path, contents string) {
t.Helper()
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
@@ -461,3 +1022,8 @@ func mustWriteFile(t *testing.T, path, contents string) {
t.Fatalf("WriteFile(%q): %v", path, err)
}
}
func boolPtr(v bool) *bool {
p := v
return &p
}

View File

@@ -0,0 +1,88 @@
package app
import (
"fmt"
"log/slog"
"os"
"path/filepath"
"regexp"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
var envVarNamePattern = regexp.MustCompile(`^[A-Za-z_][A-Za-z0-9_]*$`)
type secretsLoadStats struct {
Dir string
Loaded int
PreservedExisting int
Skipped int
}
func loadSecretsFromConfig(cfg *config.Config, logger *slog.Logger) (*secretsLoadStats, error) {
if cfg == nil || cfg.Pipeline == nil || cfg.Pipeline.Secrets == nil {
return nil, nil
}
rawDir := strings.TrimSpace(cfg.Pipeline.Secrets.EnvDir)
if rawDir == "" {
return nil, nil
}
resolvedDir := rawDir
if !filepath.IsAbs(resolvedDir) {
cwd, err := os.Getwd()
if err != nil {
return nil, fmt.Errorf("resolve secrets env_dir %q from current working directory: %w", rawDir, err)
}
resolvedDir = filepath.Join(cwd, resolvedDir)
}
resolvedDir = filepath.Clean(resolvedDir)
entries, err := os.ReadDir(resolvedDir)
if err != nil {
return nil, fmt.Errorf("read secrets env_dir %q: %w", resolvedDir, err)
}
stats := &secretsLoadStats{Dir: resolvedDir}
for _, entry := range entries {
name := entry.Name()
if !envVarNamePattern.MatchString(name) {
stats.Skipped++
continue
}
if entry.IsDir() {
stats.Skipped++
continue
}
secretPath := filepath.Join(resolvedDir, name)
bytes, err := os.ReadFile(secretPath)
if err != nil {
return nil, fmt.Errorf("read secret file %q: %w", secretPath, err)
}
value := strings.TrimRight(string(bytes), "\r\n")
if _, exists := os.LookupEnv(name); exists {
stats.PreservedExisting++
continue
}
if err := os.Setenv(name, value); err != nil {
return nil, fmt.Errorf("set environment variable %q from %q: %w", name, secretPath, err)
}
stats.Loaded++
}
if logger != nil {
logger.Info(
"loaded secret environment variables from filesystem",
"secrets_env_dir", stats.Dir,
"loaded", stats.Loaded,
"preserved_existing", stats.PreservedExisting,
"skipped", stats.Skipped,
)
}
return stats, nil
}

View File

@@ -0,0 +1,155 @@
package app
import (
"os"
"path/filepath"
"runtime"
"strings"
"testing"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func TestLoadSecretsFromConfigLoadsValidFiles(t *testing.T) {
dir := t.TempDir()
mustWriteSecretFile(t, filepath.Join(dir, "NARRATIO_TEST_SECRET_A"), "value-1\n")
mustWriteSecretFile(t, filepath.Join(dir, "NARRATIO_TEST_SECRET_B"), "value-2\r\n")
mustWriteSecretFile(t, filepath.Join(dir, "not-valid-name.txt"), "ignored")
cfg := &config.Config{
Pipeline: &config.PipelineConfig{
Secrets: &config.SecretsConfig{EnvDir: dir},
},
}
stats, err := loadSecretsFromConfig(cfg, nil)
if err != nil {
t.Fatalf("loadSecretsFromConfig() error = %v", err)
}
if stats == nil {
t.Fatal("stats = nil, want non-nil")
}
if stats.Loaded != 2 {
t.Fatalf("Loaded = %d, want 2", stats.Loaded)
}
if stats.PreservedExisting != 0 {
t.Fatalf("PreservedExisting = %d, want 0", stats.PreservedExisting)
}
if stats.Skipped == 0 {
t.Fatalf("Skipped = %d, want > 0 for invalid filename", stats.Skipped)
}
if got := os.Getenv("NARRATIO_TEST_SECRET_A"); got != "value-1" {
t.Fatalf("NARRATIO_TEST_SECRET_A = %q, want value-1", got)
}
if got := os.Getenv("NARRATIO_TEST_SECRET_B"); got != "value-2" {
t.Fatalf("NARRATIO_TEST_SECRET_B = %q, want value-2", got)
}
}
func TestLoadSecretsFromConfigPreservesExistingEnv(t *testing.T) {
t.Setenv("OBJECT_STORAGE_KEY", "existing")
dir := t.TempDir()
mustWriteSecretFile(t, filepath.Join(dir, "OBJECT_STORAGE_KEY"), "from-file\n")
cfg := &config.Config{
Pipeline: &config.PipelineConfig{
Secrets: &config.SecretsConfig{EnvDir: dir},
},
}
stats, err := loadSecretsFromConfig(cfg, nil)
if err != nil {
t.Fatalf("loadSecretsFromConfig() error = %v", err)
}
if stats.PreservedExisting != 1 {
t.Fatalf("PreservedExisting = %d, want 1", stats.PreservedExisting)
}
if got := os.Getenv("OBJECT_STORAGE_KEY"); got != "existing" {
t.Fatalf("OBJECT_STORAGE_KEY = %q, want existing", got)
}
}
func TestLoadSecretsFromConfigRelativeDirUsesCWD(t *testing.T) {
cwd := t.TempDir()
secretsDir := filepath.Join(cwd, "secrets")
if err := os.MkdirAll(secretsDir, 0o755); err != nil {
t.Fatalf("MkdirAll(%q): %v", secretsDir, err)
}
mustWriteSecretFile(t, filepath.Join(secretsDir, "OBJECT_STORAGE_KEY_ID"), "id-123\n")
originalWD, err := os.Getwd()
if err != nil {
t.Fatalf("Getwd(): %v", err)
}
if err := os.Chdir(cwd); err != nil {
t.Fatalf("Chdir(%q): %v", cwd, err)
}
t.Cleanup(func() {
_ = os.Chdir(originalWD)
})
cfg := &config.Config{
Pipeline: &config.PipelineConfig{
Secrets: &config.SecretsConfig{EnvDir: "./secrets"},
},
}
if _, err := loadSecretsFromConfig(cfg, nil); err != nil {
t.Fatalf("loadSecretsFromConfig() error = %v", err)
}
if got := os.Getenv("OBJECT_STORAGE_KEY_ID"); got != "id-123" {
t.Fatalf("OBJECT_STORAGE_KEY_ID = %q, want id-123", got)
}
}
func TestLoadSecretsFromConfigMissingDirFails(t *testing.T) {
cfg := &config.Config{
Pipeline: &config.PipelineConfig{
Secrets: &config.SecretsConfig{EnvDir: filepath.Join(t.TempDir(), "missing")},
},
}
_, err := loadSecretsFromConfig(cfg, nil)
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "read secrets env_dir") {
t.Fatalf("error = %q, want read-dir context", err.Error())
}
}
func TestLoadSecretsFromConfigUnreadableValidEntryFails(t *testing.T) {
if runtime.GOOS == "windows" {
t.Skip("symlink behavior differs on windows")
}
dir := t.TempDir()
broken := filepath.Join(dir, "OPENROUTER_API_KEY")
if err := os.Symlink(filepath.Join(dir, "does-not-exist"), broken); err != nil {
t.Fatalf("Symlink(%q): %v", broken, err)
}
cfg := &config.Config{
Pipeline: &config.PipelineConfig{
Secrets: &config.SecretsConfig{EnvDir: dir},
},
}
_, err := loadSecretsFromConfig(cfg, nil)
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "read secret file") {
t.Fatalf("error = %q, want read secret file context", err.Error())
}
}
func mustWriteSecretFile(t *testing.T, path, contents string) {
t.Helper()
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
t.Fatalf("MkdirAll(%q): %v", path, err)
}
if err := os.WriteFile(path, []byte(contents), 0o600); err != nil {
t.Fatalf("WriteFile(%q): %v", path, err)
}
}

View File

@@ -0,0 +1,86 @@
package app
import (
"bytes"
"context"
"os"
"path/filepath"
"strings"
"testing"
)
func TestPlanUsesDiscoveredSessionTemplateWithSessionID(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
sessionTemplate := `session_id: "{{ session_id }}"
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
if err := os.WriteFile(sessionPath, []byte(sessionTemplate), 0o644); err != nil {
t.Fatalf("write session template: %v", err)
}
cwd := filepath.Dir(sessionPath)
originalWD, err := os.Getwd()
if err != nil {
t.Fatalf("Getwd(): %v", err)
}
if err := os.Chdir(cwd); err != nil {
t.Fatalf("Chdir(%q): %v", cwd, err)
}
t.Cleanup(func() { _ = os.Chdir(originalWD) })
var out bytes.Buffer
if err := Plan(context.Background(), []string{"--config", pipelinePath, "--session-id", "2026-04-04"}, &out); err != nil {
t.Fatalf("Plan() error = %v", err)
}
if !strings.Contains(out.String(), "narratio plan: workdir prepared") {
t.Fatalf("output = %q, want plan output", out.String())
}
}
func TestPlanFailsWhenSessionIDMismatchesConcreteSession(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var out bytes.Buffer
err := Plan(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath, "--session-id", "2026-04-04"}, &out)
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "session_id mismatch") {
t.Fatalf("error = %q, want mismatch context", err.Error())
}
}
func TestRunStageAcceptsSessionIDFlagAndParsesStageName(t *testing.T) {
workspaceRoot := t.TempDir()
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
var out bytes.Buffer
err := RunStage(context.Background(), []string{"--config", pipelinePath, "--session", sessionPath, "--session-id", "2026-05-03", "prepare"}, &out)
if err != nil {
t.Fatalf("RunStage() error = %v", err)
}
if !strings.Contains(out.String(), "stage=prepare") {
t.Fatalf("output = %q, want stage output", out.String())
}
}
func TestResolveSessionConfigPathErrorIncludesSearchedPaths(t *testing.T) {
_, err := resolveSessionConfigPathWithCandidates("", []string{"./session.yml", "/usr/local/etc/narratio/session.yml", "/etc/narratio/session.yml"})
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "searched") {
t.Fatalf("error = %q, want searched paths", err.Error())
}
if !strings.Contains(err.Error(), "pass --session") {
t.Fatalf("error = %q, want explicit-session guidance", err.Error())
}
}

View File

@@ -0,0 +1,49 @@
package app
import (
"errors"
"fmt"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
func resolveSessionConfigPath(flagValue string) (string, error) {
return resolveSessionConfigPathWithCandidates(flagValue, config.DefaultSessionConfigSearchPaths)
}
func resolveSessionConfigPathWithCandidates(flagValue string, candidates []string) (string, error) {
if explicit := strings.TrimSpace(flagValue); explicit != "" {
return explicit, nil
}
ordered := make([]string, 0, len(candidates))
for _, raw := range candidates {
path := strings.TrimSpace(raw)
if path == "" {
continue
}
ordered = append(ordered, path)
info, err := os.Stat(path)
if err == nil {
if info.IsDir() {
continue
}
return filepath.Clean(path), nil
}
if errors.Is(err, os.ErrNotExist) {
continue
}
return "", fmt.Errorf("check default session config %q: %w", path, err)
}
if len(ordered) == 0 {
return "", fmt.Errorf("no session config path provided and no default locations configured")
}
return "", fmt.Errorf(
"no session config path provided and no default session config found; searched: %s; pass --session to use an explicit path",
strings.Join(ordered, ", "),
)
}

View File

@@ -0,0 +1,68 @@
package app
import (
"os"
"path/filepath"
"strings"
"testing"
)
func TestResolveSessionConfigPathWithCandidatesExplicitWins(t *testing.T) {
got, err := resolveSessionConfigPathWithCandidates(" ./custom/session.yml ", []string{"./session.yml", "/a", "/b"})
if err != nil {
t.Fatalf("resolveSessionConfigPathWithCandidates() error = %v", err)
}
if got != "./custom/session.yml" {
t.Fatalf("resolved path = %q, want explicit path", got)
}
}
func TestResolveSessionConfigPathWithCandidatesUsesFirstExisting(t *testing.T) {
dir := t.TempDir()
first := filepath.Join(dir, "first.yml")
second := filepath.Join(dir, "second.yml")
if err := os.WriteFile(second, []byte("session_id: 2026-05-03\n"), 0o644); err != nil {
t.Fatalf("write second default: %v", err)
}
got, err := resolveSessionConfigPathWithCandidates("", []string{first, second})
if err != nil {
t.Fatalf("resolveSessionConfigPathWithCandidates() error = %v", err)
}
if got != filepath.Clean(second) {
t.Fatalf("resolved path = %q, want %q", got, filepath.Clean(second))
}
}
func TestResolveSessionConfigPathWithCandidatesPrecedence(t *testing.T) {
dir := t.TempDir()
first := filepath.Join(dir, "first.yml")
second := filepath.Join(dir, "second.yml")
if err := os.WriteFile(first, []byte("session_id: 2026-05-03\n"), 0o644); err != nil {
t.Fatalf("write first default: %v", err)
}
if err := os.WriteFile(second, []byte("session_id: 2026-05-03\n"), 0o644); err != nil {
t.Fatalf("write second default: %v", err)
}
got, err := resolveSessionConfigPathWithCandidates("", []string{first, second})
if err != nil {
t.Fatalf("resolveSessionConfigPathWithCandidates() error = %v", err)
}
if got != filepath.Clean(first) {
t.Fatalf("resolved path = %q, want first candidate %q", got, filepath.Clean(first))
}
}
func TestResolveSessionConfigPathWithCandidatesMissing(t *testing.T) {
_, err := resolveSessionConfigPathWithCandidates("", []string{"/does/not/exist/one.yml", "/does/not/exist/two.yml"})
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "no default session config found") {
t.Fatalf("error = %q, want missing-defaults context", err.Error())
}
if !strings.Contains(err.Error(), "pass --session") {
t.Fatalf("error = %q, want explicit-path guidance", err.Error())
}
}

View File

@@ -63,7 +63,7 @@ func TestExecuteStagesDefaultWiringUsesWhisperXHTTPClient(t *testing.T) {
t.Fatal("audio file payload was empty")
}
outPath := filepath.Join(cfg.Pipeline.Workspace.Root, "work", cfg.Session.SessionID, "transcripts", "raw", "alice.json")
outPath := filepath.Join(cfg.Pipeline.Workspace.Root, "work", cfg.Session.Campaign, cfg.Session.SessionID, "transcripts", "raw", "alice.json")
data, err := os.ReadFile(outPath)
if err != nil {
t.Fatalf("ReadFile(%q) error = %v", outPath, err)

View File

@@ -0,0 +1,313 @@
package artifacts
import (
"encoding/json"
"errors"
"fmt"
"os"
"path/filepath"
"regexp"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
const (
ArtifactTranscriptMerged = "narratio.transcript.merged"
ArtifactTranscriptPolished = "narratio.transcript.polished"
ArtifactTranscriptFull = "narratio.transcript.full"
ArtifactTranscriptTrimmed = "narratio.transcript.trimmed"
ArtifactBoundsSession = "narratio.bounds.session"
)
// ErrSessionArtifactNotFound is returned when no readable artifact exists for a known ID.
var ErrSessionArtifactNotFound = errors.New("session artifact not found")
var configuredArtifactSourceRE = regexp.MustCompile(`^narratio\.artifact\.[a-z][a-z0-9_]*$`)
type artifactContentKind string
const (
contentTranscriptJSON artifactContentKind = "transcript_json"
contentJSON artifactContentKind = "json"
contentText artifactContentKind = "text"
)
type artifactSpec struct {
ID string
CanonicalRelPath string
ProducerStage string
OutputKind string
ContentKind artifactContentKind
}
var artifactRegistry = map[string]artifactSpec{
ArtifactTranscriptMerged: {
ID: ArtifactTranscriptMerged,
CanonicalRelPath: "transcripts/merged.json",
ProducerStage: "merge",
OutputKind: "transcript_merged",
ContentKind: contentTranscriptJSON,
},
ArtifactTranscriptPolished: {
ID: ArtifactTranscriptPolished,
CanonicalRelPath: "transcripts/processed.json",
ProducerStage: "polish",
OutputKind: "transcript_processed",
ContentKind: contentTranscriptJSON,
},
ArtifactTranscriptFull: {
ID: ArtifactTranscriptFull,
CanonicalRelPath: "transcripts/normalized.json",
ProducerStage: "normalize",
OutputKind: "transcript_normalized",
ContentKind: contentTranscriptJSON,
},
ArtifactTranscriptTrimmed: {
ID: ArtifactTranscriptTrimmed,
CanonicalRelPath: "transcripts/trimmed.json",
ProducerStage: "trim",
OutputKind: "transcript_trimmed",
ContentKind: contentTranscriptJSON,
},
ArtifactBoundsSession: {
ID: ArtifactBoundsSession,
CanonicalRelPath: "artifacts/session_bounds.json",
ProducerStage: "trim",
OutputKind: "session_bounds",
ContentKind: contentJSON,
},
}
// ResolvedSessionArtifact describes one session-level artifact lookup result.
type ResolvedSessionArtifact struct {
ID string
Path string
ProducerStage string
OutputKind string
ProducerRunID string
Provenance string
}
// SessionArtifactNotFoundError includes context when a known artifact cannot be read.
type SessionArtifactNotFoundError struct {
ArtifactID string
}
func (e *SessionArtifactNotFoundError) Error() string {
return fmt.Sprintf("%s: %q", ErrSessionArtifactNotFound, e.ArtifactID)
}
func (e *SessionArtifactNotFoundError) Unwrap() error {
return ErrSessionArtifactNotFound
}
// NormalizeSessionArtifactSource validates canonical artifact IDs.
func NormalizeSessionArtifactSource(source string) (string, error) {
normalized := strings.TrimSpace(source)
if normalized == "" {
return "", fmt.Errorf("artifact source is required")
}
if _, ok := artifactRegistry[normalized]; !ok {
return "", fmt.Errorf("unsupported artifact source %q", source)
}
return normalized, nil
}
// IsConfiguredArtifactSource returns true when source is narratio.artifact.<name>.
func IsConfiguredArtifactSource(source string) bool {
return configuredArtifactSourceRE.MatchString(strings.TrimSpace(source))
}
// ResolveSessionArtifact resolves a symbolic source to a readable local session artifact path.
// Resolution order is manifest producer outputs first, then canonical session path fallback.
func ResolveSessionArtifact(paths SessionPaths, m *manifest.Manifest, source string) (ResolvedSessionArtifact, error) {
id, err := NormalizeSessionArtifactSource(source)
if err != nil {
return ResolvedSessionArtifact{}, err
}
spec := artifactRegistry[id]
for _, candidate := range manifestArtifactCandidates(paths, m, spec) {
exists, isDir, statErr := pathExists(candidate.Path)
if statErr != nil {
return ResolvedSessionArtifact{}, fmt.Errorf("stat %q: %w", candidate.Path, statErr)
}
if !exists || isDir {
continue
}
resolved := candidate
resolved.ID = spec.ID
resolved.ProducerStage = spec.ProducerStage
resolved.OutputKind = spec.OutputKind
if err := validateResolvedContent(resolved.Path, spec.ContentKind); err != nil {
return ResolvedSessionArtifact{}, fmt.Errorf("validate %q: %w", resolved.ID, err)
}
return resolved, nil
}
fallbackPath := filepath.Join(paths.Root, filepath.FromSlash(spec.CanonicalRelPath))
exists, isDir, statErr := pathExists(fallbackPath)
if statErr != nil {
return ResolvedSessionArtifact{}, fmt.Errorf("stat %q: %w", fallbackPath, statErr)
}
if exists && !isDir {
if err := validateResolvedContent(fallbackPath, spec.ContentKind); err != nil {
return ResolvedSessionArtifact{}, fmt.Errorf("validate %q: %w", spec.ID, err)
}
return ResolvedSessionArtifact{
ID: spec.ID,
Path: filepath.Clean(fallbackPath),
ProducerStage: spec.ProducerStage,
OutputKind: spec.OutputKind,
Provenance: "fallback.canonical_path",
}, nil
}
return ResolvedSessionArtifact{}, &SessionArtifactNotFoundError{ArtifactID: spec.ID}
}
// ResolveSessionArtifactWithCatalog resolves built-in sources using existing rules and resolves
// configured narratio.artifact.<name> sources through runtime catalog availability.
func ResolveSessionArtifactWithCatalog(paths SessionPaths, m *manifest.Manifest, source string, catalog *ArtifactCatalog) (ResolvedSessionArtifact, error) {
normalized := strings.TrimSpace(source)
if !IsConfiguredArtifactSource(normalized) {
return ResolveSessionArtifact(paths, m, normalized)
}
if catalog == nil {
return ResolvedSessionArtifact{}, fmt.Errorf("configured artifact source %q requires runtime artifact catalog", source)
}
entry, ok := catalog.Lookup(normalized)
if !ok {
return ResolvedSessionArtifact{}, fmt.Errorf("unsupported artifact source %q", source)
}
if !entry.Available {
return ResolvedSessionArtifact{}, &SessionArtifactNotFoundError{ArtifactID: normalized}
}
if err := validateResolvedContent(entry.Path, contentText); err != nil {
return ResolvedSessionArtifact{}, fmt.Errorf("validate %q: %w", normalized, err)
}
return ResolvedSessionArtifact{
ID: normalized,
Path: filepath.Clean(entry.Path),
ProducerStage: entry.ProducerStage,
OutputKind: entry.OutputKind,
Provenance: entry.Provenance,
}, nil
}
func manifestArtifactCandidates(paths SessionPaths, m *manifest.Manifest, spec artifactSpec) []ResolvedSessionArtifact {
if m == nil || len(m.Stages) == 0 || spec.ProducerStage == "" || spec.OutputKind == "" {
return nil
}
sr := m.Stages[spec.ProducerStage]
if sr == nil {
return nil
}
candidates := make([]ResolvedSessionArtifact, 0, len(sr.Outputs))
for _, out := range sr.Outputs {
if strings.TrimSpace(out.Kind) != spec.OutputKind {
continue
}
if strings.TrimSpace(out.LocalPath) == "" {
continue
}
resolved := filepath.Clean(ResolveSessionLocalPathForRead(paths, out.LocalPath))
if resolved == "" {
continue
}
candidates = append(candidates, ResolvedSessionArtifact{
Path: resolved,
ProducerRunID: strings.TrimSpace(out.ProducerRunID),
Provenance: "manifest." + spec.ProducerStage + ".outputs",
})
}
return dedupeResolvedArtifacts(candidates)
}
func dedupeResolvedArtifacts(values []ResolvedSessionArtifact) []ResolvedSessionArtifact {
seen := map[string]struct{}{}
out := make([]ResolvedSessionArtifact, 0, len(values))
for _, value := range values {
key := filepath.Clean(strings.TrimSpace(value.Path))
if key == "" {
continue
}
if _, ok := seen[key]; ok {
continue
}
seen[key] = struct{}{}
value.Path = key
out = append(out, value)
}
return out
}
func pathExists(path string) (exists bool, isDir bool, err error) {
info, err := os.Stat(path)
if err == nil {
return true, info.IsDir(), nil
}
if errors.Is(err, os.ErrNotExist) {
return false, false, nil
}
return false, false, err
}
func validateResolvedContent(path string, kind artifactContentKind) error {
switch kind {
case contentTranscriptJSON:
return validateTranscriptSegmentsJSON(path)
case contentJSON:
return validateJSONContent(path)
case contentText:
return validateNonEmptyContent(path)
default:
return fmt.Errorf("unsupported content kind %q", kind)
}
}
func validateTranscriptSegmentsJSON(path string) error {
data, err := os.ReadFile(path)
if err != nil {
return fmt.Errorf("read file: %w", err)
}
var payload map[string]any
if err := json.Unmarshal(data, &payload); err != nil {
return fmt.Errorf("decode json: %w", err)
}
segments, ok := payload["segments"]
if !ok {
return fmt.Errorf("top-level segments is required")
}
if _, ok := segments.([]any); !ok {
return fmt.Errorf("top-level segments must be an array")
}
return nil
}
func validateJSONContent(path string) error {
data, err := os.ReadFile(path)
if err != nil {
return fmt.Errorf("read file: %w", err)
}
var payload any
if err := json.Unmarshal(data, &payload); err != nil {
return fmt.Errorf("decode json: %w", err)
}
return nil
}
func validateNonEmptyContent(path string) error {
info, err := os.Stat(path)
if err != nil {
return fmt.Errorf("stat file: %w", err)
}
if info.IsDir() {
return fmt.Errorf("path is a directory")
}
if info.Size() <= 0 {
return fmt.Errorf("file is empty")
}
return nil
}

View File

@@ -0,0 +1,268 @@
package artifacts
import (
"errors"
"os"
"path/filepath"
"strings"
"testing"
"time"
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
)
func TestNormalizeSessionArtifactSource(t *testing.T) {
tests := []struct {
name string
source string
wantID string
wantErr string
}{
{name: "legacy alias processed unsupported", source: "processed_transcript", wantErr: "unsupported artifact source"},
{name: "legacy alias normalized unsupported", source: "normalized_transcript", wantErr: "unsupported artifact source"},
{name: "legacy alias trimmed unsupported", source: "trimmed_transcript", wantErr: "unsupported artifact source"},
{name: "configured source unsupported in built-in normalization", source: "narratio.artifact.session_recap", wantErr: "unsupported artifact source"},
{name: "canonical", source: ArtifactTranscriptTrimmed, wantID: ArtifactTranscriptTrimmed},
{name: "unsupported", source: "narratio.unknown", wantErr: "unsupported artifact source"},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
got, err := NormalizeSessionArtifactSource(tt.source)
if tt.wantErr != "" {
if err == nil || !strings.Contains(err.Error(), tt.wantErr) {
t.Fatalf("NormalizeSessionArtifactSource() error = %v, want contains %q", err, tt.wantErr)
}
return
}
if err != nil {
t.Fatalf("NormalizeSessionArtifactSource() error = %v", err)
}
if got != tt.wantID {
t.Fatalf("NormalizeSessionArtifactSource() = %q, want %q", got, tt.wantID)
}
})
}
}
func TestResolveSessionArtifactPrefersManifestOutput(t *testing.T) {
workspace := t.TempDir()
paths := buildSessionPaths(workspace, "campaign", "session")
if err := os.MkdirAll(paths.ArtifactsDir, 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
manifestPath := filepath.Join(paths.ArtifactsDir, "normalized.from-manifest.json")
if err := os.WriteFile(manifestPath, []byte(`{"segments":[]}`), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
canonicalPath := filepath.Join(paths.TranscriptsDir, "normalized.json")
if err := os.MkdirAll(filepath.Dir(canonicalPath), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
if err := os.WriteFile(canonicalPath, []byte(`{"segments":[{"id":123}]}`), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
m := manifest.New("session", time.Now().UTC())
m.MarkStageSucceeded("normalize", time.Now().UTC(), []manifest.ArtifactRecord{
{Kind: "transcript_normalized", LocalPath: manifestPath, ProducerRunID: "run-123"},
})
resolved, err := ResolveSessionArtifact(paths, m, ArtifactTranscriptFull)
if err != nil {
t.Fatalf("ResolveSessionArtifact() error = %v", err)
}
if resolved.Path != manifestPath {
t.Fatalf("resolved path = %q, want %q", resolved.Path, manifestPath)
}
if resolved.Provenance != "manifest.normalize.outputs" {
t.Fatalf("provenance = %q, want manifest.normalize.outputs", resolved.Provenance)
}
if resolved.ProducerRunID != "run-123" {
t.Fatalf("producer run id = %q, want run-123", resolved.ProducerRunID)
}
}
func TestResolveSessionArtifactFallsBackToCanonicalPath(t *testing.T) {
workspace := t.TempDir()
paths := buildSessionPaths(workspace, "campaign", "session")
canonicalPath := filepath.Join(paths.TranscriptsDir, "trimmed.json")
if err := os.MkdirAll(filepath.Dir(canonicalPath), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
if err := os.WriteFile(canonicalPath, []byte(`{"segments":[]}`), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
resolved, err := ResolveSessionArtifact(paths, nil, ArtifactTranscriptTrimmed)
if err != nil {
t.Fatalf("ResolveSessionArtifact() error = %v", err)
}
if resolved.Path != canonicalPath {
t.Fatalf("resolved path = %q, want %q", resolved.Path, canonicalPath)
}
if resolved.Provenance != "fallback.canonical_path" {
t.Fatalf("provenance = %q, want fallback.canonical_path", resolved.Provenance)
}
}
func TestResolveSessionArtifactMissingReturnsTypedError(t *testing.T) {
workspace := t.TempDir()
paths := buildSessionPaths(workspace, "campaign", "session")
_, err := ResolveSessionArtifact(paths, nil, ArtifactTranscriptTrimmed)
if err == nil {
t.Fatal("expected error, got nil")
}
if !errors.Is(err, ErrSessionArtifactNotFound) {
t.Fatalf("errors.Is(err, ErrSessionArtifactNotFound) = false; err=%v", err)
}
}
func TestResolveSessionArtifactValidatesTranscriptShape(t *testing.T) {
workspace := t.TempDir()
paths := buildSessionPaths(workspace, "campaign", "session")
canonicalPath := filepath.Join(paths.TranscriptsDir, "processed.json")
if err := os.MkdirAll(filepath.Dir(canonicalPath), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
if err := os.WriteFile(canonicalPath, []byte(`{"not_segments":[]}`), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
_, err := ResolveSessionArtifact(paths, nil, ArtifactTranscriptPolished)
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "top-level segments is required") {
t.Fatalf("error = %q, want segments validation error", err.Error())
}
}
func TestResolveSessionArtifactWithCatalogBuiltInBehaviorUnchanged(t *testing.T) {
workspace := t.TempDir()
paths := buildSessionPaths(workspace, "campaign", "session")
canonicalPath := filepath.Join(paths.TranscriptsDir, "trimmed.json")
if err := os.MkdirAll(filepath.Dir(canonicalPath), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
if err := os.WriteFile(canonicalPath, []byte(`{"segments":[]}`), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
resolved, err := ResolveSessionArtifactWithCatalog(paths, nil, ArtifactTranscriptTrimmed, NewArtifactCatalog())
if err != nil {
t.Fatalf("ResolveSessionArtifactWithCatalog() error = %v", err)
}
if resolved.Path != canonicalPath {
t.Fatalf("resolved.Path = %q, want %q", resolved.Path, canonicalPath)
}
if resolved.Provenance != "fallback.canonical_path" {
t.Fatalf("provenance = %q, want fallback.canonical_path", resolved.Provenance)
}
}
func TestResolveSessionArtifactWithCatalogConfiguredAvailableGenerated(t *testing.T) {
workspace := t.TempDir()
paths := buildSessionPaths(workspace, "campaign", "session")
outputPath := filepath.Join(paths.ArtifactsDir, "session_recap.md")
if err := os.MkdirAll(filepath.Dir(outputPath), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
if err := os.WriteFile(outputPath, []byte("recap\n"), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
catalog := NewArtifactCatalog()
if err := catalog.RegisterConfiguredArtifacts(
map[string]ConfiguredArtifactDefinition{
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
},
nil,
); err != nil {
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
}
sourceID := ConfiguredArtifactSourceID("session_recap")
if err := catalog.MarkAvailableGenerated(sourceID, outputPath); err != nil {
t.Fatalf("MarkAvailableGenerated() error = %v", err)
}
resolved, err := ResolveSessionArtifactWithCatalog(paths, nil, sourceID, catalog)
if err != nil {
t.Fatalf("ResolveSessionArtifactWithCatalog() error = %v", err)
}
if resolved.Path != outputPath {
t.Fatalf("resolved.Path = %q, want %q", resolved.Path, outputPath)
}
if resolved.Provenance != ArtifactProvenanceGeneratedCurrentAnalyzeRun {
t.Fatalf("provenance = %q, want %q", resolved.Provenance, ArtifactProvenanceGeneratedCurrentAnalyzeRun)
}
}
func TestResolveSessionArtifactWithCatalogConfiguredAvailableFromDisk(t *testing.T) {
workspace := t.TempDir()
paths := buildSessionPaths(workspace, "campaign", "session")
outputPath := filepath.Join(paths.ArtifactsDir, "player_handout.md")
if err := os.MkdirAll(filepath.Dir(outputPath), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
if err := os.WriteFile(outputPath, []byte("handout\n"), 0o644); err != nil {
t.Fatalf("WriteFile() error = %v", err)
}
catalog := NewArtifactCatalog()
if err := catalog.RegisterConfiguredArtifacts(
map[string]ConfiguredArtifactDefinition{
"player_handout": {Enabled: false, OutputPath: "artifacts/player_handout.md"},
},
nil,
); err != nil {
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
}
sourceID := ConfiguredArtifactSourceID("player_handout")
if err := catalog.MarkAvailableFromDisk(sourceID, outputPath); err != nil {
t.Fatalf("MarkAvailableFromDisk() error = %v", err)
}
resolved, err := ResolveSessionArtifactWithCatalog(paths, nil, sourceID, catalog)
if err != nil {
t.Fatalf("ResolveSessionArtifactWithCatalog() error = %v", err)
}
if resolved.Provenance != ArtifactProvenanceDisabledFromDisk {
t.Fatalf("provenance = %q, want %q", resolved.Provenance, ArtifactProvenanceDisabledFromDisk)
}
}
func TestResolveSessionArtifactWithCatalogConfiguredPlannedButUnavailable(t *testing.T) {
workspace := t.TempDir()
paths := buildSessionPaths(workspace, "campaign", "session")
catalog := NewArtifactCatalog()
if err := catalog.RegisterConfiguredArtifacts(
map[string]ConfiguredArtifactDefinition{
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
},
nil,
); err != nil {
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
}
_, err := ResolveSessionArtifactWithCatalog(paths, nil, ConfiguredArtifactSourceID("session_recap"), catalog)
if err == nil {
t.Fatal("expected error, got nil")
}
if !errors.Is(err, ErrSessionArtifactNotFound) {
t.Fatalf("errors.Is(err, ErrSessionArtifactNotFound)=false; err=%v", err)
}
}
func TestResolveSessionArtifactWithCatalogUnsupportedConfiguredSourceFails(t *testing.T) {
workspace := t.TempDir()
paths := buildSessionPaths(workspace, "campaign", "session")
catalog := NewArtifactCatalog()
_, err := ResolveSessionArtifactWithCatalog(paths, nil, "narratio.artifact.unknown", catalog)
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "unsupported artifact source") {
t.Fatalf("error = %q, want unsupported artifact source", err.Error())
}
}

View File

@@ -0,0 +1,224 @@
package artifacts
import (
"fmt"
"sort"
"strings"
)
const (
ArtifactProvenanceGeneratedCurrentAnalyzeRun = "generated.current_analyze_run"
ArtifactProvenanceDisabledFromDisk = "filesystem.disabled_artifact_output"
)
// ConfiguredArtifactDefinition describes one configured analyze artifact.
type ConfiguredArtifactDefinition struct {
Enabled bool
OutputPath string
}
// CatalogEntry is one runtime catalog entry resolved by source ID.
type CatalogEntry struct {
SourceID string
ConfiguredKey string
CanonicalRelPath string
ProducerStage string
OutputKind string
Planned bool
Executable bool
Available bool
Path string
Provenance string
}
// ArtifactCatalog tracks built-in and configured artifact definitions and runtime state.
type ArtifactCatalog struct {
entries map[string]CatalogEntry
configuredIndex map[string]string
}
// NewArtifactCatalog returns an empty runtime artifact catalog.
func NewArtifactCatalog() *ArtifactCatalog {
return &ArtifactCatalog{
entries: map[string]CatalogEntry{},
configuredIndex: map[string]string{},
}
}
// ConfiguredArtifactSourceID converts a configured artifact key into canonical source ID.
func ConfiguredArtifactSourceID(key string) string {
return "narratio.artifact." + strings.TrimSpace(key)
}
// RegisterBuiltIns registers built-in source definitions used by runtime artifact resolution.
func (c *ArtifactCatalog) RegisterBuiltIns() error {
for _, id := range runtimeBuiltInArtifactIDs() {
spec, ok := artifactRegistry[id]
if !ok {
return fmt.Errorf("register built-ins: source %q not found in artifact registry", id)
}
if err := c.addEntry(CatalogEntry{
SourceID: spec.ID,
CanonicalRelPath: spec.CanonicalRelPath,
ProducerStage: spec.ProducerStage,
OutputKind: spec.OutputKind,
Planned: true,
}); err != nil {
return fmt.Errorf("register built-ins: %w", err)
}
}
return nil
}
// RegisterConfiguredArtifacts registers configured artifacts and applies executable selection.
func (c *ArtifactCatalog) RegisterConfiguredArtifacts(
configured map[string]ConfiguredArtifactDefinition,
selected []string,
) error {
keys := make([]string, 0, len(configured))
for key := range configured {
keys = append(keys, key)
}
sort.Strings(keys)
selectedSet := map[string]struct{}{}
for _, key := range selected {
trimmed := strings.TrimSpace(key)
if trimmed == "" {
return fmt.Errorf("selected artifact keys must be non-empty")
}
selectedSet[trimmed] = struct{}{}
}
for _, key := range keys {
trimmed := strings.TrimSpace(key)
if trimmed == "" {
return fmt.Errorf("configured artifact keys must be non-empty")
}
def := configured[key]
sourceID := ConfiguredArtifactSourceID(trimmed)
if _, exists := c.configuredIndex[trimmed]; exists {
return fmt.Errorf("duplicate configured artifact key %q", trimmed)
}
executable := def.Enabled
if len(selectedSet) > 0 {
_, executable = selectedSet[trimmed]
}
if err := c.addEntry(CatalogEntry{
SourceID: sourceID,
ConfiguredKey: trimmed,
CanonicalRelPath: strings.TrimSpace(def.OutputPath),
ProducerStage: "analyze",
OutputKind: "scriptorium_artifact",
Planned: true,
Executable: executable,
}); err != nil {
return fmt.Errorf("register configured artifact %q: %w", trimmed, err)
}
c.configuredIndex[trimmed] = sourceID
}
if len(selectedSet) > 0 {
for key := range selectedSet {
if _, ok := c.configuredIndex[key]; !ok {
return fmt.Errorf("selected artifact %q is not configured", key)
}
}
}
return nil
}
// Lookup returns one catalog entry by source ID.
func (c *ArtifactCatalog) Lookup(sourceID string) (CatalogEntry, bool) {
if c == nil {
return CatalogEntry{}, false
}
entry, ok := c.entries[strings.TrimSpace(sourceID)]
return entry, ok
}
// SourceIDForConfiguredKey returns canonical source ID for one configured key.
func (c *ArtifactCatalog) SourceIDForConfiguredKey(key string) (string, bool) {
if c == nil {
return "", false
}
sourceID, ok := c.configuredIndex[strings.TrimSpace(key)]
return sourceID, ok
}
// ListConfigured returns configured entries sorted by configured key.
func (c *ArtifactCatalog) ListConfigured() []CatalogEntry {
if c == nil || len(c.configuredIndex) == 0 {
return nil
}
keys := make([]string, 0, len(c.configuredIndex))
for key := range c.configuredIndex {
keys = append(keys, key)
}
sort.Strings(keys)
out := make([]CatalogEntry, 0, len(keys))
for _, key := range keys {
sourceID := c.configuredIndex[key]
out = append(out, c.entries[sourceID])
}
return out
}
// MarkAvailableGenerated marks one source as available in current analyze execution.
func (c *ArtifactCatalog) MarkAvailableGenerated(sourceID, path string) error {
return c.markAvailable(sourceID, path, ArtifactProvenanceGeneratedCurrentAnalyzeRun)
}
// MarkAvailableFromDisk marks one source as available from disabled artifact on disk.
func (c *ArtifactCatalog) MarkAvailableFromDisk(sourceID, path string) error {
return c.markAvailable(sourceID, path, ArtifactProvenanceDisabledFromDisk)
}
func (c *ArtifactCatalog) markAvailable(sourceID, path, provenance string) error {
if c == nil {
return fmt.Errorf("artifact catalog is nil")
}
normalizedID := strings.TrimSpace(sourceID)
entry, ok := c.entries[normalizedID]
if !ok {
return fmt.Errorf("unknown artifact source %q", sourceID)
}
trimmedPath := strings.TrimSpace(path)
if trimmedPath == "" {
return fmt.Errorf("artifact path is required")
}
entry.Available = true
entry.Path = trimmedPath
entry.Provenance = provenance
c.entries[normalizedID] = entry
return nil
}
func (c *ArtifactCatalog) addEntry(entry CatalogEntry) error {
if c == nil {
return fmt.Errorf("artifact catalog is nil")
}
sourceID := strings.TrimSpace(entry.SourceID)
if sourceID == "" {
return fmt.Errorf("source id is required")
}
if _, exists := c.entries[sourceID]; exists {
return fmt.Errorf("source id %q is already registered", sourceID)
}
entry.SourceID = sourceID
c.entries[sourceID] = entry
return nil
}
func runtimeBuiltInArtifactIDs() []string {
return []string{
ArtifactTranscriptMerged,
ArtifactTranscriptPolished,
ArtifactTranscriptFull,
ArtifactTranscriptTrimmed,
ArtifactBoundsSession,
}
}

View File

@@ -0,0 +1,188 @@
package artifacts
import "testing"
func TestArtifactCatalogRegisterBuiltInsAndLookup(t *testing.T) {
catalog := NewArtifactCatalog()
if err := catalog.RegisterBuiltIns(); err != nil {
t.Fatalf("RegisterBuiltIns() error = %v", err)
}
entry, ok := catalog.Lookup(ArtifactTranscriptFull)
if !ok {
t.Fatalf("Lookup(%q) ok = false, want true", ArtifactTranscriptFull)
}
if !entry.Planned {
t.Fatalf("entry.Planned = false, want true")
}
if entry.Executable {
t.Fatalf("entry.Executable = true, want false")
}
if entry.CanonicalRelPath != "transcripts/normalized.json" {
t.Fatalf("entry.CanonicalRelPath = %q, want transcripts/normalized.json", entry.CanonicalRelPath)
}
}
func TestArtifactCatalogRegisterConfiguredArtifactsDefaultsToEnabled(t *testing.T) {
catalog := NewArtifactCatalog()
if err := catalog.RegisterConfiguredArtifacts(map[string]ConfiguredArtifactDefinition{
"player_handout": {Enabled: false, OutputPath: "artifacts/player_handout.md"},
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
}, nil); err != nil {
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
}
recapID, ok := catalog.SourceIDForConfiguredKey("session_recap")
if !ok {
t.Fatal("SourceIDForConfiguredKey(session_recap) ok = false, want true")
}
recap, ok := catalog.Lookup(recapID)
if !ok {
t.Fatalf("Lookup(%q) ok = false, want true", recapID)
}
if !recap.Executable {
t.Fatalf("recap.Executable = false, want true")
}
handoutID, ok := catalog.SourceIDForConfiguredKey("player_handout")
if !ok {
t.Fatal("SourceIDForConfiguredKey(player_handout) ok = false, want true")
}
handout, ok := catalog.Lookup(handoutID)
if !ok {
t.Fatalf("Lookup(%q) ok = false, want true", handoutID)
}
if handout.Executable {
t.Fatalf("handout.Executable = true, want false")
}
}
func TestArtifactCatalogRegisterConfiguredArtifactsSelectedSetOverridesEnabled(t *testing.T) {
catalog := NewArtifactCatalog()
if err := catalog.RegisterConfiguredArtifacts(
map[string]ConfiguredArtifactDefinition{
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
"player_handout": {Enabled: false, OutputPath: "artifacts/player_handout.md"},
},
[]string{"player_handout"},
); err != nil {
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
}
entries := catalog.ListConfigured()
if len(entries) != 2 {
t.Fatalf("ListConfigured() len = %d, want 2", len(entries))
}
if entries[0].ConfiguredKey != "player_handout" || entries[0].Executable != true {
t.Fatalf("entries[0] = %+v, want player_handout executable", entries[0])
}
if entries[1].ConfiguredKey != "session_recap" || entries[1].Executable != false {
t.Fatalf("entries[1] = %+v, want session_recap disabled by selection", entries[1])
}
}
func TestArtifactCatalogRejectsSelectedUnknownArtifact(t *testing.T) {
catalog := NewArtifactCatalog()
err := catalog.RegisterConfiguredArtifacts(
map[string]ConfiguredArtifactDefinition{
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
},
[]string{"unknown"},
)
if err == nil {
t.Fatal("RegisterConfiguredArtifacts() error = nil, want non-nil")
}
}
func TestArtifactCatalogRejectsConfiguredSourceConflictAcrossRegistrations(t *testing.T) {
catalog := NewArtifactCatalog()
if err := catalog.RegisterConfiguredArtifacts(
map[string]ConfiguredArtifactDefinition{
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
},
nil,
); err != nil {
t.Fatalf("RegisterConfiguredArtifacts() first call error = %v", err)
}
err := catalog.RegisterConfiguredArtifacts(
map[string]ConfiguredArtifactDefinition{
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
},
nil,
)
if err == nil {
t.Fatal("RegisterConfiguredArtifacts() error = nil, want non-nil")
}
}
func TestArtifactCatalogMarkAvailableGenerated(t *testing.T) {
catalog := NewArtifactCatalog()
if err := catalog.RegisterConfiguredArtifacts(
map[string]ConfiguredArtifactDefinition{
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
},
nil,
); err != nil {
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
}
sourceID, _ := catalog.SourceIDForConfiguredKey("session_recap")
if err := catalog.MarkAvailableGenerated(sourceID, "/tmp/session_recap.md"); err != nil {
t.Fatalf("MarkAvailableGenerated() error = %v", err)
}
entry, _ := catalog.Lookup(sourceID)
if !entry.Available {
t.Fatalf("entry.Available = false, want true")
}
if entry.Provenance != ArtifactProvenanceGeneratedCurrentAnalyzeRun {
t.Fatalf("entry.Provenance = %q, want %q", entry.Provenance, ArtifactProvenanceGeneratedCurrentAnalyzeRun)
}
}
func TestArtifactCatalogMarkAvailableFromDisk(t *testing.T) {
catalog := NewArtifactCatalog()
if err := catalog.RegisterConfiguredArtifacts(
map[string]ConfiguredArtifactDefinition{
"session_recap": {Enabled: false, OutputPath: "artifacts/session_recap.md"},
},
nil,
); err != nil {
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
}
sourceID, _ := catalog.SourceIDForConfiguredKey("session_recap")
if err := catalog.MarkAvailableFromDisk(sourceID, "/tmp/session_recap.md"); err != nil {
t.Fatalf("MarkAvailableFromDisk() error = %v", err)
}
entry, _ := catalog.Lookup(sourceID)
if !entry.Available {
t.Fatalf("entry.Available = false, want true")
}
if entry.Provenance != ArtifactProvenanceDisabledFromDisk {
t.Fatalf("entry.Provenance = %q, want %q", entry.Provenance, ArtifactProvenanceDisabledFromDisk)
}
}
func TestArtifactCatalogLookupPlannedButUnavailable(t *testing.T) {
catalog := NewArtifactCatalog()
if err := catalog.RegisterConfiguredArtifacts(
map[string]ConfiguredArtifactDefinition{
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
},
nil,
); err != nil {
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
}
sourceID, _ := catalog.SourceIDForConfiguredKey("session_recap")
entry, ok := catalog.Lookup(sourceID)
if !ok {
t.Fatalf("Lookup(%q) ok = false, want true", sourceID)
}
if !entry.Planned {
t.Fatalf("entry.Planned = false, want true")
}
if entry.Available {
t.Fatalf("entry.Available = true, want false")
}
}

View File

@@ -30,13 +30,13 @@ func NewLocalStore(workspaceRoot string) *LocalStore {
return &LocalStore{WorkspaceRoot: workspaceRoot}
}
// SessionPaths resolves canonical paths for a session workdir.
func (s *LocalStore) SessionPaths(sessionID string) SessionPaths {
return buildSessionPaths(s.WorkspaceRoot, sessionID)
// SessionPathsFor resolves canonical campaign-aware paths for a session workdir.
func (s *LocalStore) SessionPathsFor(campaign, sessionID string) SessionPaths {
return buildSessionPaths(s.WorkspaceRoot, campaign, sessionID)
}
// EnsureLayout creates and verifies the canonical session workdir directory layout.
func (s *LocalStore) EnsureLayout(sessionID string) (SessionPaths, error) {
// EnsureLayoutFor creates and verifies campaign-aware session layout.
func (s *LocalStore) EnsureLayoutFor(campaign, sessionID string) (SessionPaths, error) {
if strings.TrimSpace(s.WorkspaceRoot) == "" {
return SessionPaths{}, fmt.Errorf("workspace root is required")
}
@@ -44,7 +44,22 @@ func (s *LocalStore) EnsureLayout(sessionID string) (SessionPaths, error) {
return SessionPaths{}, fmt.Errorf("sessionID is required")
}
paths := s.SessionPaths(sessionID)
campaign = strings.TrimSpace(campaign)
if campaign == "" {
return SessionPaths{}, fmt.Errorf("campaign is required")
}
return s.ensureLayout(s.SessionPathsFor(campaign, sessionID))
}
func (s *LocalStore) ensureLayout(paths SessionPaths) (SessionPaths, error) {
if strings.TrimSpace(s.WorkspaceRoot) == "" {
return SessionPaths{}, fmt.Errorf("workspace root is required")
}
if strings.TrimSpace(paths.SessionID) == "" {
return SessionPaths{}, fmt.Errorf("sessionID is required")
}
dirs := []string{
paths.Root,
paths.InputsDir,
@@ -53,8 +68,11 @@ func (s *LocalStore) EnsureLayout(sessionID string) (SessionPaths, error) {
paths.TranscriptsRawDir,
paths.TranscriptsTrimmedDir,
paths.ArtifactsDir,
paths.ReportsDir,
paths.ConfigDir,
paths.LogsDir,
paths.CurrentDir,
paths.RunsDir,
}
for _, dir := range dirs {
@@ -66,13 +84,16 @@ func (s *LocalStore) EnsureLayout(sessionID string) (SessionPaths, error) {
return paths, nil
}
// CopyInput copies an input file into the session workdir under destRelativePath.
func (s *LocalStore) CopyInput(sessionID, srcPath, destRelativePath string) (Ref, error) {
paths, err := s.EnsureLayout(sessionID)
// CopyInputFor copies an input file into the campaign-aware session workdir under destRelativePath.
func (s *LocalStore) CopyInputFor(campaign, sessionID, srcPath, destRelativePath string) (Ref, error) {
paths, err := s.EnsureLayoutFor(campaign, sessionID)
if err != nil {
return Ref{}, err
}
return s.copyInputWithPaths(paths, sessionID, srcPath, destRelativePath)
}
func (s *LocalStore) copyInputWithPaths(paths SessionPaths, sessionID, srcPath, destRelativePath string) (Ref, error) {
destAbs, err := resolveInRoot(paths.Root, destRelativePath)
if err != nil {
return Ref{}, fmt.Errorf("copy input: %w", err)
@@ -173,13 +194,16 @@ func (s *LocalStore) Checksum(path string) (string, error) {
return digest, nil
}
// AcquireSessionLock acquires an exclusive lock file for a session workdir.
func (s *LocalStore) AcquireSessionLock(sessionID string) (*LockHandle, error) {
paths, err := s.EnsureLayout(sessionID)
// AcquireSessionLockFor acquires an exclusive lock file for a campaign/session workdir.
func (s *LocalStore) AcquireSessionLockFor(campaign, sessionID string) (*LockHandle, error) {
paths, err := s.EnsureLayoutFor(campaign, sessionID)
if err != nil {
return nil, err
}
return s.acquireSessionLockForPaths(paths)
}
func (s *LocalStore) acquireSessionLockForPaths(paths SessionPaths) (*LockHandle, error) {
f, err := os.OpenFile(paths.LockPath, os.O_CREATE|os.O_EXCL|os.O_WRONLY, 0o644)
if err != nil {
if errors.Is(err, os.ErrExist) {

View File

@@ -10,9 +10,9 @@ import (
func TestEnsureLayoutCreatesExpectedDirectories(t *testing.T) {
store := NewLocalStore(t.TempDir())
paths, err := store.EnsureLayout("session-1")
paths, err := store.EnsureLayoutFor("sample-campaign", "session-1")
if err != nil {
t.Fatalf("EnsureLayout() error = %v", err)
t.Fatalf("EnsureLayoutFor() error = %v", err)
}
checkDirExists(t, paths.Root)
@@ -22,8 +22,11 @@ func TestEnsureLayoutCreatesExpectedDirectories(t *testing.T) {
checkDirExists(t, paths.TranscriptsRawDir)
checkDirExists(t, paths.TranscriptsTrimmedDir)
checkDirExists(t, paths.ArtifactsDir)
checkDirExists(t, paths.ReportsDir)
checkDirExists(t, paths.ConfigDir)
checkDirExists(t, paths.LogsDir)
checkDirExists(t, paths.CurrentDir)
checkDirExists(t, paths.RunsDir)
if filepath.Base(paths.ManifestPath) != "manifest.json" {
t.Fatalf("ManifestPath = %q, want basename manifest.json", paths.ManifestPath)
@@ -33,6 +36,17 @@ func TestEnsureLayoutCreatesExpectedDirectories(t *testing.T) {
}
}
func TestEnsureLayoutForRequiresCampaign(t *testing.T) {
store := NewLocalStore(t.TempDir())
_, err := store.EnsureLayoutFor("", "session-1")
if err == nil {
t.Fatal("expected campaign-required error, got nil")
}
if !strings.Contains(err.Error(), "campaign is required") {
t.Fatalf("error = %v, want campaign-required error", err)
}
}
func TestChecksumCalculation(t *testing.T) {
store := NewLocalStore(t.TempDir())
path := filepath.Join(t.TempDir(), "sample.txt")
@@ -53,9 +67,9 @@ func TestChecksumCalculation(t *testing.T) {
func TestLockAcquireRelease(t *testing.T) {
store := NewLocalStore(t.TempDir())
lock, err := store.AcquireSessionLock("session-1")
lock, err := store.AcquireSessionLockFor("sample-campaign", "session-1")
if err != nil {
t.Fatalf("AcquireSessionLock() error = %v", err)
t.Fatalf("AcquireSessionLockFor() error = %v", err)
}
exists, err := store.Exists(lock.path)
@@ -81,15 +95,15 @@ func TestLockAcquireRelease(t *testing.T) {
func TestLockConflict(t *testing.T) {
store := NewLocalStore(t.TempDir())
lock1, err := store.AcquireSessionLock("session-1")
lock1, err := store.AcquireSessionLockFor("sample-campaign", "session-1")
if err != nil {
t.Fatalf("first AcquireSessionLock() error = %v", err)
t.Fatalf("first AcquireSessionLockFor() error = %v", err)
}
defer func() {
_ = store.ReleaseSessionLock(lock1)
}()
_, err = store.AcquireSessionLock("session-1")
_, err = store.AcquireSessionLockFor("sample-campaign", "session-1")
if err == nil {
t.Fatal("expected lock conflict error, got nil")
}
@@ -138,9 +152,9 @@ func TestCopyInput(t *testing.T) {
t.Fatalf("WriteFile() error = %v", err)
}
ref, err := store.CopyInput("session-1", srcPath, "inputs/speakers.yml")
ref, err := store.CopyInputFor("sample-campaign", "session-1", srcPath, "inputs/speakers.yml")
if err != nil {
t.Fatalf("CopyInput() error = %v", err)
t.Fatalf("CopyInputFor() error = %v", err)
}
if ref.Kind != "input" {

View File

@@ -1,10 +1,16 @@
package artifacts
import "path/filepath"
import (
"path/filepath"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
// SessionPaths contains canonical local paths for one session work directory.
type SessionPaths struct {
WorkspaceRoot string
CampaignID string
SessionID string
Root string
InputsDir string
AudioDir string
@@ -12,32 +18,73 @@ type SessionPaths struct {
TranscriptsRawDir string
TranscriptsTrimmedDir string
ArtifactsDir string
ReportsDir string
ConfigDir string
LogsDir string
CurrentDir string
RunsDir string
ManifestPath string
LockPath string
}
// SessionWorkDir returns the work directory for one session.
func SessionWorkDir(rootDir, sessionID string) string {
return filepath.Join(rootDir, "work", sessionID)
// SessionWorkDirForCampaign returns the canonical campaign-aware work directory for one session.
func SessionWorkDirForCampaign(rootDir, campaign, sessionID string) string {
return filepath.Join(rootDir, config.PathWorkDirSegment, campaign, sessionID)
}
func buildSessionPaths(workspaceRoot, sessionID string) SessionPaths {
root := SessionWorkDir(workspaceRoot, sessionID)
transcripts := filepath.Join(root, "transcripts")
// SessionManifestPathForCampaign returns the canonical session manifest path.
func SessionManifestPathForCampaign(rootDir, campaign, sessionID string) string {
return filepath.Join(SessionWorkDirForCampaign(rootDir, campaign, sessionID), config.PathManifestFile)
}
// SessionRunsDirForCampaign returns the canonical runs directory for one session.
func SessionRunsDirForCampaign(rootDir, campaign, sessionID string) string {
return filepath.Join(SessionWorkDirForCampaign(rootDir, campaign, sessionID), config.PathRunsDirSegment)
}
// SessionRunRootForCampaign returns the canonical run root under runs/{run_id}.
func SessionRunRootForCampaign(rootDir, campaign, sessionID, runID string) string {
return filepath.Join(SessionRunsDirForCampaign(rootDir, campaign, sessionID), runID)
}
// SessionRunManifestPathForCampaign returns the canonical run manifest path under runs/{run_id}/manifest.json.
func SessionRunManifestPathForCampaign(rootDir, campaign, sessionID, runID string) string {
return filepath.Join(SessionRunRootForCampaign(rootDir, campaign, sessionID, runID), config.PathManifestFile)
}
// SessionRunStageDirForCampaign returns the canonical stage directory under runs/{run_id}/{stage}.
func SessionRunStageDirForCampaign(rootDir, campaign, sessionID, runID, stageName string) string {
return filepath.Join(SessionRunRootForCampaign(rootDir, campaign, sessionID, runID), stageName)
}
// SessionSpoolAudioDir returns the campaign/session/run scoped local spool audio path.
func SessionSpoolAudioDir(spoolRoot, campaign, sessionID, runID string) string {
return filepath.Join(spoolRoot, campaign, sessionID, runID, config.PathAudioDirSegment)
}
func buildSessionPaths(workspaceRoot, campaign, sessionID string) SessionPaths {
root := SessionWorkDirForCampaign(workspaceRoot, campaign, sessionID)
return buildSessionPathsFromRoot(workspaceRoot, campaign, sessionID, root)
}
func buildSessionPathsFromRoot(workspaceRoot, campaign, sessionID, root string) SessionPaths {
return SessionPaths{
WorkspaceRoot: workspaceRoot,
CampaignID: campaign,
SessionID: sessionID,
Root: root,
InputsDir: filepath.Join(root, "inputs"),
AudioDir: filepath.Join(root, "audio"),
TranscriptsDir: transcripts,
TranscriptsRawDir: filepath.Join(transcripts, "raw"),
TranscriptsTrimmedDir: filepath.Join(transcripts, "trimmed"),
ArtifactsDir: filepath.Join(root, "artifacts"),
ConfigDir: filepath.Join(root, "config"),
LogsDir: filepath.Join(root, "logs"),
ManifestPath: filepath.Join(root, "manifest.json"),
LockPath: filepath.Join(root, ".lock"),
InputsDir: filepath.Join(root, config.PathInputsDirSegment),
AudioDir: filepath.Join(root, config.PathAudioDirSegment),
TranscriptsDir: filepath.Join(root, config.PathTranscriptsSegment),
TranscriptsRawDir: filepath.Join(root, filepath.FromSlash(config.PathTranscriptsRaw)),
TranscriptsTrimmedDir: filepath.Join(root, filepath.FromSlash(config.PathTranscriptsTrimmed)),
ArtifactsDir: filepath.Join(root, config.PathArtifactsDirSegment),
ReportsDir: filepath.Join(root, config.PathReportsDirSegment),
ConfigDir: filepath.Join(root, config.PathConfigDirSegment),
LogsDir: filepath.Join(root, config.PathLogsDirSegment),
CurrentDir: filepath.Join(root, config.PathCurrentDirSegment),
RunsDir: filepath.Join(root, config.PathRunsDirSegment),
ManifestPath: filepath.Join(root, config.PathManifestFile),
LockPath: filepath.Join(root, config.PathLockFile),
}
}

View File

@@ -0,0 +1,59 @@
package artifacts
import (
"path/filepath"
"testing"
)
func TestSessionWorkDirForCampaign(t *testing.T) {
root := "/tmp/workspace"
got := SessionWorkDirForCampaign(root, "forsaken", "2026-04-19")
want := filepath.Join(root, "work", "forsaken", "2026-04-19")
if got != want {
t.Fatalf("SessionWorkDirForCampaign() = %q, want %q", got, want)
}
}
func TestSessionManifestPathForCampaign(t *testing.T) {
root := "/tmp/workspace"
got := SessionManifestPathForCampaign(root, "forsaken", "2026-04-19")
want := filepath.Join(root, "work", "forsaken", "2026-04-19", "manifest.json")
if got != want {
t.Fatalf("SessionManifestPathForCampaign() = %q, want %q", got, want)
}
}
func TestSessionRunRootAndStageDirForCampaign(t *testing.T) {
root := "/tmp/workspace"
runID := "20260515T031522Z-a1b2c3d4"
runRoot := SessionRunRootForCampaign(root, "forsaken", "2026-04-19", runID)
wantRoot := filepath.Join(root, "work", "forsaken", "2026-04-19", "runs", runID)
if runRoot != wantRoot {
t.Fatalf("SessionRunRootForCampaign() = %q, want %q", runRoot, wantRoot)
}
stageDir := SessionRunStageDirForCampaign(root, "forsaken", "2026-04-19", runID, "transcribe")
wantStage := filepath.Join(root, "work", "forsaken", "2026-04-19", "runs", runID, "transcribe")
if stageDir != wantStage {
t.Fatalf("SessionRunStageDirForCampaign() = %q, want %q", stageDir, wantStage)
}
}
func TestSessionRunManifestPathForCampaign(t *testing.T) {
root := "/tmp/workspace"
runID := "20260515T031522Z-a1b2c3d4"
got := SessionRunManifestPathForCampaign(root, "forsaken", "2026-04-19", runID)
want := filepath.Join(root, "work", "forsaken", "2026-04-19", "runs", runID, "manifest.json")
if got != want {
t.Fatalf("SessionRunManifestPathForCampaign() = %q, want %q", got, want)
}
}
func TestSessionSpoolAudioDir(t *testing.T) {
root := "/var/spool/narratio"
got := SessionSpoolAudioDir(root, "forsaken", "2026-04-19", "20260515T031522Z-a1b2c3d4")
want := filepath.Join(root, "forsaken", "2026-04-19", "20260515T031522Z-a1b2c3d4", "audio")
if got != want {
t.Fatalf("SessionSpoolAudioDir() = %q, want %q", got, want)
}
}

View File

@@ -8,7 +8,7 @@ import (
func TestResolveSessionLocalPathForRead(t *testing.T) {
workspace := t.TempDir()
paths := buildSessionPaths(workspace, "s-1")
paths := buildSessionPaths(workspace, "sample-campaign", "s-1")
if err := os.MkdirAll(paths.TranscriptsRawDir, 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
}
@@ -41,7 +41,7 @@ func TestResolveSessionLocalPathForReadRelativeWorkspaceRootQualifiedPath(t *tes
t.Fatalf("Rel() error = %v", err)
}
paths := buildSessionPaths(workspaceRel, "s-1")
paths := buildSessionPaths(workspaceRel, "sample-campaign", "s-1")
target := filepath.Join(paths.TranscriptsRawDir, "alice.json")
if err := os.MkdirAll(filepath.Dir(target), 0o755); err != nil {
t.Fatalf("MkdirAll() error = %v", err)
@@ -50,7 +50,7 @@ func TestResolveSessionLocalPathForReadRelativeWorkspaceRootQualifiedPath(t *tes
t.Fatalf("WriteFile() error = %v", err)
}
manifestPath := filepath.Join(workspaceRel, "work", "s-1", "transcripts", "raw", "alice.json")
manifestPath := filepath.Join(workspaceRel, "work", "sample-campaign", "s-1", "transcripts", "raw", "alice.json")
got := ResolveSessionLocalPathForRead(paths, manifestPath)
if got != filepath.Clean(manifestPath) {
t.Fatalf("resolution = %q, want %q", got, filepath.Clean(manifestPath))

View File

@@ -0,0 +1,30 @@
package artifacts
import (
"crypto/rand"
"encoding/hex"
"fmt"
"io"
"time"
)
// NewRunID returns a run ID in format: YYYYMMDDTHHMMSSZ-xxxxxxxx.
func NewRunID() (string, error) {
return NewRunIDWith(time.Now().UTC(), rand.Reader)
}
// NewRunIDWith returns a run ID in format: YYYYMMDDTHHMMSSZ-xxxxxxxx
// using an injected timestamp and randomness source.
func NewRunIDWith(now time.Time, random io.Reader) (string, error) {
if random == nil {
random = rand.Reader
}
var suffix [4]byte
if _, err := io.ReadFull(random, suffix[:]); err != nil {
return "", fmt.Errorf("generate run id random suffix: %w", err)
}
ts := now.UTC().Format("20060102T150405Z")
return ts + "-" + hex.EncodeToString(suffix[:]), nil
}

View File

@@ -0,0 +1,37 @@
package artifacts
import (
"bytes"
"regexp"
"strings"
"testing"
"time"
)
func TestNewRunIDWithFormat(t *testing.T) {
now := time.Date(2026, 5, 15, 3, 15, 22, 0, time.UTC)
random := bytes.NewReader([]byte{0xa1, 0xb2, 0xc3, 0xd4})
runID, err := NewRunIDWith(now, random)
if err != nil {
t.Fatalf("NewRunIDWith() error = %v", err)
}
if runID != "20260515T031522Z-a1b2c3d4" {
t.Fatalf("runID = %q, want %q", runID, "20260515T031522Z-a1b2c3d4")
}
}
func TestNewRunIDWithShape(t *testing.T) {
runID, err := NewRunIDWith(time.Now().UTC(), bytes.NewReader([]byte{0x01, 0x02, 0x03, 0x04}))
if err != nil {
t.Fatalf("NewRunIDWith() error = %v", err)
}
pattern := regexp.MustCompile(`^\d{8}T\d{6}Z-[0-9a-f]{8}$`)
if !pattern.MatchString(runID) {
t.Fatalf("runID = %q, want pattern %q", runID, pattern.String())
}
suffix := runID[len(runID)-8:]
if strings.ToLower(suffix) != suffix {
t.Fatalf("runID suffix = %q, want lowercase", suffix)
}
}

View File

@@ -11,19 +11,19 @@ type S3Store struct {
Prefix string
}
// SessionPaths is not implemented for S3-backed storage.
func (s *S3Store) SessionPaths(_ string) SessionPaths {
// SessionPathsFor is not implemented for S3-backed storage.
func (s *S3Store) SessionPathsFor(_, _ string) SessionPaths {
return SessionPaths{}
}
// EnsureLayout returns a not-yet-implemented error in the scaffold.
func (s *S3Store) EnsureLayout(_ string) (SessionPaths, error) {
return SessionPaths{}, fmt.Errorf("artifacts s3 ensure layout: not yet implemented")
// EnsureLayoutFor returns a not-yet-implemented error in the scaffold.
func (s *S3Store) EnsureLayoutFor(_, _ string) (SessionPaths, error) {
return SessionPaths{}, fmt.Errorf("artifacts s3 ensure layout for campaign/session: not yet implemented")
}
// CopyInput returns a not-yet-implemented error in the scaffold.
func (s *S3Store) CopyInput(_, _, _ string) (Ref, error) {
return Ref{}, fmt.Errorf("artifacts s3 copy input: not yet implemented")
// CopyInputFor returns a not-yet-implemented error in the scaffold.
func (s *S3Store) CopyInputFor(_, _, _, _ string) (Ref, error) {
return Ref{}, fmt.Errorf("artifacts s3 copy input for campaign/session: not yet implemented")
}
// Exists returns a not-yet-implemented error in the scaffold.
@@ -46,9 +46,9 @@ func (s *S3Store) Checksum(_ string) (string, error) {
return "", fmt.Errorf("artifacts s3 checksum: not yet implemented")
}
// AcquireSessionLock returns a not-yet-implemented error in the scaffold.
func (s *S3Store) AcquireSessionLock(_ string) (*LockHandle, error) {
return nil, fmt.Errorf("artifacts s3 acquire lock: not yet implemented")
// AcquireSessionLockFor returns a not-yet-implemented error in the scaffold.
func (s *S3Store) AcquireSessionLockFor(_, _ string) (*LockHandle, error) {
return nil, fmt.Errorf("artifacts s3 acquire lock for campaign/session: not yet implemented")
}
// ReleaseSessionLock returns a not-yet-implemented error in the scaffold.

View File

@@ -0,0 +1,75 @@
package artifacts
import (
"path"
"strings"
"gitea.maximumdirect.net/eric/narratio/internal/config"
)
// S3SessionPrefix builds the canonical S3 session prefix.
// Format: {root_prefix}/campaigns/{campaign}/sessions/{session_id}/
func S3SessionPrefix(rootPrefix, campaign, sessionID string) string {
prefix := path.Join(
cleanS3PathPart(rootPrefix),
config.S3CampaignsSegment,
cleanS3PathPart(campaign),
config.S3SessionsSegment,
cleanS3PathPart(sessionID),
)
return ensureS3TrailingSlash(prefix)
}
// S3RunPrefix builds the canonical S3 run prefix.
// Format: {session_prefix}/runs/{run_id}/
func S3RunPrefix(sessionPrefix, runID string) string {
prefix := path.Join(strings.TrimSuffix(cleanS3Key(sessionPrefix), "/"), config.S3RunsSegment, cleanS3PathPart(runID))
return ensureS3TrailingSlash(prefix)
}
// S3AudioPrefix builds the session audio prefix from configured audio_s3.prefix.
// Format: {session_prefix}/{audio_s3.prefix}
func S3AudioPrefix(sessionPrefix, audioPrefix string) string {
key := path.Join(strings.TrimSuffix(cleanS3Key(sessionPrefix), "/"), cleanS3Key(audioPrefix))
return ensureS3TrailingSlash(key)
}
// S3CurrentManifestKey returns the current manifest pointer key.
// Format: {session_prefix}/current/manifest.json
func S3CurrentManifestKey(sessionPrefix string) string {
return path.Join(strings.TrimSuffix(cleanS3Key(sessionPrefix), "/"), config.S3CurrentSegment, config.S3ManifestFile)
}
// S3CurrentRunPointerKey returns the current run pointer key.
// Format: {session_prefix}/current/run_id.txt
func S3CurrentRunPointerKey(sessionPrefix string) string {
return path.Join(strings.TrimSuffix(cleanS3Key(sessionPrefix), "/"), config.S3CurrentSegment, config.S3RunIDFile)
}
// S3PromotedArtifactKey returns the destination key for one promoted artifact.
// Format: {session_prefix}/{promotion.to}
func S3PromotedArtifactKey(sessionPrefix, to string) string {
return path.Join(strings.TrimSuffix(cleanS3Key(sessionPrefix), "/"), cleanS3Key(to))
}
// S3RunRelativeDestinationKey returns a run-scoped key for a workdir-relative path.
// Format: {run_prefix}/{relative_workdir_path}
func S3RunRelativeDestinationKey(runPrefix, relativeWorkdirPath string) string {
return path.Join(strings.TrimSuffix(cleanS3Key(runPrefix), "/"), cleanS3Key(relativeWorkdirPath))
}
func ensureS3TrailingSlash(v string) string {
key := cleanS3Key(v)
if key == "" {
return ""
}
return strings.TrimSuffix(key, "/") + "/"
}
func cleanS3PathPart(v string) string {
return strings.Trim(strings.ReplaceAll(strings.TrimSpace(v), "\\", "/"), "/")
}
func cleanS3Key(v string) string {
return strings.ReplaceAll(strings.TrimSpace(v), "\\", "/")
}

View File

@@ -0,0 +1,45 @@
package artifacts
import (
"strings"
"testing"
)
func TestS3KeyConstruction(t *testing.T) {
runID := "20260515T031522Z-a1b2c3d4"
sessionPrefix := S3SessionPrefix("dnd", "forsaken", "2026-04-19")
if sessionPrefix != "dnd/campaigns/forsaken/sessions/2026-04-19/" {
t.Fatalf("sessionPrefix = %q", sessionPrefix)
}
audioPrefix := S3AudioPrefix(sessionPrefix, "audio/")
if audioPrefix != "dnd/campaigns/forsaken/sessions/2026-04-19/audio/" {
t.Fatalf("audioPrefix = %q", audioPrefix)
}
runPrefix := S3RunPrefix(sessionPrefix, runID)
wantRunPrefix := "dnd/campaigns/forsaken/sessions/2026-04-19/runs/" + runID + "/"
if runPrefix != wantRunPrefix {
t.Fatalf("runPrefix = %q, want %q", runPrefix, wantRunPrefix)
}
runPointer := S3CurrentRunPointerKey(sessionPrefix)
if runPointer != "dnd/campaigns/forsaken/sessions/2026-04-19/current/run_id.txt" {
t.Fatalf("run pointer key = %q", runPointer)
}
manifestKey := S3CurrentManifestKey(sessionPrefix)
if manifestKey != "dnd/campaigns/forsaken/sessions/2026-04-19/current/manifest.json" {
t.Fatalf("manifest key = %q", manifestKey)
}
promoted := S3PromotedArtifactKey(sessionPrefix, "transcripts/trimmed.json")
if promoted != "dnd/campaigns/forsaken/sessions/2026-04-19/transcripts/trimmed.json" {
t.Fatalf("promoted key = %q", promoted)
}
runRelative := S3RunRelativeDestinationKey(runPrefix, `logs\whisperx.stdout.log`)
if !strings.HasSuffix(runRelative, "/logs/whisperx.stdout.log") {
t.Fatalf("runRelative key = %q, want normalized forward slashes", runRelative)
}
}

View File

@@ -15,13 +15,13 @@ type Ref struct {
// Store is the local artifact/workdir abstraction used by orchestration code.
type Store interface {
SessionPaths(sessionID string) SessionPaths
EnsureLayout(sessionID string) (SessionPaths, error)
CopyInput(sessionID, srcPath, destRelativePath string) (Ref, error)
SessionPathsFor(campaign, sessionID string) SessionPaths
EnsureLayoutFor(campaign, sessionID string) (SessionPaths, error)
CopyInputFor(campaign, sessionID, srcPath, destRelativePath string) (Ref, error)
Exists(path string) (bool, error)
ExistsRef(ref Ref) (bool, error)
WriteFileAtomic(path string, data []byte, perm os.FileMode) error
Checksum(path string) (string, error)
AcquireSessionLock(sessionID string) (*LockHandle, error)
AcquireSessionLockFor(campaign, sessionID string) (*LockHandle, error)
ReleaseSessionLock(lock *LockHandle) error
}

View File

@@ -12,6 +12,9 @@ type Config struct {
type PipelineConfig struct {
Workspace WorkspaceConfig `yaml:"workspace"`
Storage StorageConfig `yaml:"storage"`
Spool SpoolConfig `yaml:"spool"`
Archive *ArchiveConfig `yaml:"archive"`
Secrets *SecretsConfig `yaml:"secrets"`
WhisperX WhisperXConfig `yaml:"whisperx"`
Seriatim SeriatimConfig `yaml:"seriatim"`
Audita AuditaConfig `yaml:"audita"`
@@ -33,14 +36,52 @@ type SessionConfig struct {
// WorkspaceConfig configures local workspace behavior.
type WorkspaceConfig struct {
Root string `yaml:"root"`
Root string `yaml:"root"`
CleanupAfterArchive bool `yaml:"cleanup_after_archive"`
}
// SecretsConfig configures optional local filesystem secret loading.
type SecretsConfig struct {
EnvDir string `yaml:"env_dir"`
}
// StorageConfig configures storage backends and related parameters.
type StorageConfig struct {
Backend string `yaml:"backend"`
Bucket string `yaml:"bucket"`
Prefix string `yaml:"prefix"`
Backend string `yaml:"backend"`
Bucket string `yaml:"bucket"`
Prefix string `yaml:"prefix"`
S3 *StorageS3Config `yaml:"s3"`
}
// StorageS3Config configures S3 storage coordinates.
type StorageS3Config struct {
Bucket string `yaml:"bucket"`
RootPrefix string `yaml:"root_prefix"`
Region string `yaml:"region"`
Endpoint string `yaml:"endpoint"`
ForcePathStyle bool `yaml:"force_path_style"`
AccessKeyIDEnv string `yaml:"access_key_id_env"`
SecretKeyEnv string `yaml:"secret_access_key_env"`
}
// SpoolConfig configures local spool storage for staged data.
type SpoolConfig struct {
Root string `yaml:"root"`
DeleteAudioAfterArchive bool `yaml:"delete_audio_after_archive"`
}
// ArchiveConfig configures archive behavior and artifact promotions.
type ArchiveConfig struct {
Enabled *bool `yaml:"enabled"`
UploadRun *bool `yaml:"upload_run"`
PromoteArtifacts []ArchivePromotionRule `yaml:"promote_artifacts"`
}
// ArchivePromotionRule configures one artifact promotion mapping.
type ArchivePromotionRule struct {
From string `yaml:"from"`
To string `yaml:"to"`
Required *bool `yaml:"required"`
}
// WhisperXConfig configures WhisperX adapter settings.
@@ -79,9 +120,14 @@ type AuditaConfig struct {
Modules []string `yaml:"modules"`
BaseURL string `yaml:"base_url"`
Model string `yaml:"model"`
LLMConcurrency *int `yaml:"llm_concurrency"`
TotalLLMConcurrency *int `yaml:"total_llm_concurrency"`
ProposalLLMConcurrency *int `yaml:"proposal_llm_concurrency"`
ValidationModel string `yaml:"validation_model"`
ValidationLLMConcurrency *int `yaml:"validation_llm_concurrency"`
TranscriptDescription string `yaml:"transcript_description"`
ConfigPath string `yaml:"config_path"`
OutputSchema string `yaml:"output_schema"`
WorkDirRetention string `yaml:"work_dir_retention"`
Report *bool `yaml:"report"`
}
@@ -130,6 +176,7 @@ type ScriptoriumConfig struct {
// ScriptoriumArtifactConfig configures one named output artifact workflow.
type ScriptoriumArtifactConfig struct {
Enabled bool `yaml:"enabled"`
DependsOn []string `yaml:"depends_on"`
RenderDebug *bool `yaml:"render_debug"`
PromptID string `yaml:"prompt_id"`
ProfileID string `yaml:"profile_id"`
@@ -169,9 +216,15 @@ type ArtifactSettings struct {
// SessionInputsConfig contains per-session input references.
type SessionInputsConfig struct {
AudioDir string `yaml:"audio_dir"`
AudioFiles []string `yaml:"audio_files"`
SpeakersFile string `yaml:"speakers_file"`
AutocorrectFile string `yaml:"autocorrect_file"`
GlossaryFile string `yaml:"glossary_file"`
AudioDir string `yaml:"audio_dir"`
AudioFiles []string `yaml:"audio_files"`
AudioS3 *SessionAudioS3Input `yaml:"audio_s3"`
SpeakersFile string `yaml:"speakers_file"`
AutocorrectFile string `yaml:"autocorrect_file"`
GlossaryFile string `yaml:"glossary_file"`
}
// SessionAudioS3Input configures S3 session-audio input discovery.
type SessionAudioS3Input struct {
Prefix string `yaml:"prefix"`
}

100
internal/config/defaults.go Normal file
View File

@@ -0,0 +1,100 @@
package config
// Default filesystem locations for pipeline configuration lookup when --config
// is omitted. Order is highest to lowest precedence.
const (
DefaultPipelineConfigPathUsrLocal = "/usr/local/etc/narratio/pipeline.yml"
DefaultPipelineConfigPathEtc = "/etc/narratio/pipeline.yml"
DefaultSessionConfigPathLocal = "./session.yml"
DefaultSessionConfigPathUsrLocal = "/usr/local/etc/narratio/session.yml"
DefaultSessionConfigPathEtc = "/etc/narratio/session.yml"
DefaultS3AccessKeyIDEnv = "OBJECT_STORAGE_KEY_ID"
DefaultS3SecretAccessKeyEnv = "OBJECT_STORAGE_KEY"
DefaultStorageS3RootPrefix = "dnd"
DefaultWorkspaceRoot = "/var/lib/narratio"
DefaultSpoolRoot = "/var/spool/narratio"
DefaultWhisperXLanguage = "en"
DefaultWhisperXTimeout = "30m"
DefaultWhisperXRetryDelay = "2s"
DefaultWhisperXConcurrency = 2
DefaultWhisperXRetries = 3
DefaultSeriatimBinary = "seriatim"
DefaultSeriatimTimeout = "10m"
DefaultSeriatimOutputSchema = "seriatim-intermediate"
DefaultSeriatimCoalesceGap = 3.0
DefaultSeriatimReport = true
DefaultAuditaBinary = "audita"
DefaultAuditaTimeout = "3h"
DefaultAuditaReport = true
DefaultScriptoriumBinary = "scriptorium"
DefaultScriptoriumTimeout = "10m"
DefaultScriptoriumArtifactOutputRoot = "artifacts"
DefaultTrimBoundsTimeout = "10m"
DefaultTrimSeriatimReport = false
DefaultNormalizeOutputPath = "transcripts/normalized.json"
DefaultNormalizeOutputSchema = "seriatim-intermediate"
DefaultNormalizeReport = true
DefaultArchiveEnabled = true
DefaultArchiveUploadRun = true
PathWorkDirSegment = "work"
PathInputsDirSegment = "inputs"
PathAudioDirSegment = "audio"
PathTranscriptsSegment = "transcripts"
PathTranscriptsRaw = "transcripts/raw"
PathTranscriptsTrimmed = "transcripts/trimmed"
PathArtifactsDirSegment = "artifacts"
PathReportsDirSegment = "reports"
PathConfigDirSegment = "config"
PathLogsDirSegment = "logs"
PathCurrentDirSegment = "current"
PathRunsDirSegment = "runs"
PathManifestFile = "manifest.json"
PathLockFile = ".lock"
PathTranscriptMerged = "transcripts/merged.json"
PathTranscriptProcessed = "transcripts/processed.json"
PathTranscriptNormalized = "transcripts/normalized.json"
PathTranscriptTrimmed = "transcripts/trimmed.json"
S3CampaignsSegment = "campaigns"
S3SessionsSegment = "sessions"
S3RunsSegment = "runs"
S3CurrentSegment = "current"
S3ManifestFile = "manifest.json"
S3RunIDFile = "run_id.txt"
)
// DefaultArchivePromoteArtifacts defines the default archive promotion rules.
// Callers should copy this slice before mutating.
var DefaultArchivePromoteArtifacts = []ArchivePromotionRule{
{From: PathTranscriptTrimmed, To: PathTranscriptTrimmed},
{From: "artifacts/session_recap.md", To: "artifacts/session_recap.md"},
}
// DefaultPipelineConfigSearchPaths defines the default search order for
// pipeline.yml when callers do not provide an explicit path.
//
// Keep this in a variable so future defaults can be extended without changing
// call sites.
var DefaultPipelineConfigSearchPaths = []string{
DefaultPipelineConfigPathUsrLocal,
DefaultPipelineConfigPathEtc,
}
// DefaultSessionConfigSearchPaths defines the default search order for
// session.yml when callers do not provide an explicit path.
//
// Keep this in a variable so future defaults can be extended without changing
// call sites.
var DefaultSessionConfigSearchPaths = []string{
DefaultSessionConfigPathLocal,
DefaultSessionConfigPathUsrLocal,
DefaultSessionConfigPathEtc,
}

View File

@@ -5,6 +5,8 @@ import (
"io"
"os"
"path/filepath"
"regexp"
"strings"
"gopkg.in/yaml.v3"
)
@@ -21,21 +23,56 @@ func LoadPipeline(path string) (*PipelineConfig, error) {
// LoadSession loads session configuration from a YAML file with strict field checking.
func LoadSession(path string) (*SessionConfig, error) {
var cfg SessionConfig
if err := decodeStrictYAML("session", path, &cfg); err != nil {
return LoadSessionWithOptions(path, SessionLoadOptions{})
}
// SessionLoadOptions configures session template rendering behavior.
type SessionLoadOptions struct {
SessionID string
}
// LoadSessionWithOptions loads session configuration from a YAML file with
// strict field checking after template rendering.
func LoadSessionWithOptions(path string, opts SessionLoadOptions) (*SessionConfig, error) {
sessionBytes, err := os.ReadFile(path)
if err != nil {
return nil, fmt.Errorf("load session config: session file %q: open: %w", path, err)
}
rendered, err := renderSessionTemplate(string(sessionBytes), opts)
if err != nil {
return nil, fmt.Errorf("load session config: %w", err)
}
var cfg SessionConfig
if err := decodeStrictYAMLFromReader("session", path, strings.NewReader(rendered), &cfg); err != nil {
return nil, fmt.Errorf("load session config: %w", err)
}
if strings.TrimSpace(opts.SessionID) != "" && strings.TrimSpace(cfg.SessionID) != "" && strings.TrimSpace(cfg.SessionID) != strings.TrimSpace(opts.SessionID) {
return nil, fmt.Errorf(
"load session config: session file %q: session_id mismatch: --session-id %q does not match rendered session_id %q",
path,
strings.TrimSpace(opts.SessionID),
strings.TrimSpace(cfg.SessionID),
)
}
return &cfg, nil
}
// Load loads and resolves combined pipeline and session configuration.
func Load(pipelinePath, sessionPath string) (*Config, error) {
return LoadWithSessionOptions(pipelinePath, sessionPath, SessionLoadOptions{})
}
// LoadWithSessionOptions loads and resolves combined pipeline and session
// configuration with session template options.
func LoadWithSessionOptions(pipelinePath, sessionPath string, sessionOpts SessionLoadOptions) (*Config, error) {
pipelineCfg, err := LoadPipeline(pipelinePath)
if err != nil {
return nil, err
}
sessionCfg, err := LoadSession(sessionPath)
sessionCfg, err := LoadSessionWithOptions(sessionPath, sessionOpts)
if err != nil {
return nil, err
}
@@ -55,7 +92,11 @@ func decodeStrictYAML(kind, path string, out any) error {
}
defer f.Close()
dec := yaml.NewDecoder(f)
return decodeStrictYAMLFromReader(kind, path, f, out)
}
func decodeStrictYAMLFromReader(kind, path string, r io.Reader, out any) error {
dec := yaml.NewDecoder(r)
dec.KnownFields(true)
if err := dec.Decode(out); err != nil {
return fmt.Errorf("%s file %q: strict decode failed: %w", kind, path, err)
@@ -69,6 +110,36 @@ func decodeStrictYAML(kind, path string, out any) error {
return nil
}
var sessionTemplatePattern = regexp.MustCompile(`\{\{\s*([a-zA-Z_][a-zA-Z0-9_]*)\s*\}\}`)
func renderSessionTemplate(content string, opts SessionLoadOptions) (string, error) {
sessionID := strings.TrimSpace(opts.SessionID)
rendered := content
if sessionID != "" {
rendered = strings.ReplaceAll(rendered, "{{session_id}}", sessionID)
rendered = strings.ReplaceAll(rendered, "{{ session_id }}", sessionID)
}
unresolved := sessionTemplatePattern.FindAllStringSubmatch(rendered, -1)
if len(unresolved) > 0 {
vars := make([]string, 0, len(unresolved))
for _, m := range unresolved {
if len(m) > 1 {
vars = append(vars, m[1])
}
}
if len(vars) > 0 {
return "", fmt.Errorf(
"session file template rendering failed: unresolved template variable(s): %s; pass --session-id when using {{ session_id }}",
strings.Join(vars, ", "),
)
}
return "", fmt.Errorf("session file template rendering failed: unresolved template placeholders remain")
}
return rendered, nil
}
func shortName(path, fallback string) string {
base := filepath.Base(path)
if base == "." || base == string(filepath.Separator) {
@@ -81,6 +152,10 @@ func applyPipelineDefaults(cfg *PipelineConfig) {
if cfg == nil {
return
}
applyWorkspaceDefaults(&cfg.Workspace)
applyStorageDefaults(&cfg.Storage)
applySpoolDefaults(&cfg.Spool)
applyArchiveDefaults(&cfg.Archive)
applyWhisperXDefaults(&cfg.WhisperX)
applySeriatimDefaults(&cfg.Seriatim)
applyAuditaDefaults(&cfg.Audita)
@@ -92,24 +167,84 @@ func applyPipelineDefaults(cfg *PipelineConfig) {
applyScriptoriumDefaults(cfg.Scriptorium)
}
func applyWorkspaceDefaults(cfg *WorkspaceConfig) {
if cfg == nil {
return
}
if cfg.Root == "" {
cfg.Root = DefaultWorkspaceRoot
}
}
func applyStorageDefaults(cfg *StorageConfig) {
if cfg == nil {
return
}
if cfg.S3 == nil {
cfg.S3 = &StorageS3Config{}
}
if cfg.S3.RootPrefix == "" {
cfg.S3.RootPrefix = DefaultStorageS3RootPrefix
}
if cfg.S3.AccessKeyIDEnv == "" {
cfg.S3.AccessKeyIDEnv = DefaultS3AccessKeyIDEnv
}
if cfg.S3.SecretKeyEnv == "" {
cfg.S3.SecretKeyEnv = DefaultS3SecretAccessKeyEnv
}
}
func applySpoolDefaults(cfg *SpoolConfig) {
if cfg == nil {
return
}
if cfg.Root == "" {
cfg.Root = DefaultSpoolRoot
}
}
func applyArchiveDefaults(cfg **ArchiveConfig) {
if cfg == nil {
return
}
if *cfg == nil {
*cfg = &ArchiveConfig{}
}
if (*cfg).Enabled == nil {
(*cfg).Enabled = boolPtr(DefaultArchiveEnabled)
}
if (*cfg).UploadRun == nil {
(*cfg).UploadRun = boolPtr(DefaultArchiveUploadRun)
}
if len((*cfg).PromoteArtifacts) == 0 {
(*cfg).PromoteArtifacts = append([]ArchivePromotionRule(nil), DefaultArchivePromoteArtifacts...)
}
for i := range (*cfg).PromoteArtifacts {
if (*cfg).PromoteArtifacts[i].Required == nil {
(*cfg).PromoteArtifacts[i].Required = boolPtr(true)
}
}
}
func applyWhisperXDefaults(cfg *WhisperXConfig) {
if cfg == nil {
return
}
if cfg.Language == "" {
cfg.Language = "en"
cfg.Language = DefaultWhisperXLanguage
}
if cfg.Timeout == "" {
cfg.Timeout = "30m"
cfg.Timeout = DefaultWhisperXTimeout
}
if cfg.RetryDelay == "" {
cfg.RetryDelay = "2s"
cfg.RetryDelay = DefaultWhisperXRetryDelay
}
if cfg.Concurrency == nil {
cfg.Concurrency = intPtr(2)
cfg.Concurrency = intPtr(DefaultWhisperXConcurrency)
}
if cfg.Retries == nil {
cfg.Retries = intPtr(3)
cfg.Retries = intPtr(DefaultWhisperXRetries)
}
}
@@ -122,17 +257,20 @@ func applySeriatimDefaults(cfg *SeriatimConfig) {
if cfg == nil {
return
}
if cfg.Binary == "" {
cfg.Binary = DefaultSeriatimBinary
}
if cfg.Timeout == "" {
cfg.Timeout = "10m"
cfg.Timeout = DefaultSeriatimTimeout
}
if cfg.OutputSchema == "" {
cfg.OutputSchema = "seriatim-intermediate"
cfg.OutputSchema = DefaultSeriatimOutputSchema
}
if cfg.CoalesceGap == nil {
cfg.CoalesceGap = float64Ptr(3.0)
cfg.CoalesceGap = float64Ptr(DefaultSeriatimCoalesceGap)
}
if cfg.Report == nil {
cfg.Report = boolPtr(true)
cfg.Report = boolPtr(DefaultSeriatimReport)
}
}
@@ -140,34 +278,14 @@ func applyAuditaDefaults(cfg *AuditaConfig) {
if cfg == nil {
return
}
if cfg.Binary == "" {
cfg.Binary = DefaultAuditaBinary
}
if cfg.Timeout == "" {
cfg.Timeout = "3h"
}
if cfg.Modules == nil {
cfg.Modules = []string{
"glossary",
"homophones",
"glossary",
"spoken_word",
"grammar",
"homophones",
"glossary",
}
}
if cfg.BaseURL == "" {
cfg.BaseURL = "https://openrouter.ai/api/v1"
}
if cfg.Model == "" {
cfg.Model = "openrouter/google/gemma-4-31b-it"
}
if cfg.LLMConcurrency == nil {
cfg.LLMConcurrency = intPtr(1)
}
if cfg.ValidationLLMConcurrency == nil {
cfg.ValidationLLMConcurrency = intPtr(1)
cfg.Timeout = DefaultAuditaTimeout
}
if cfg.Report == nil {
cfg.Report = boolPtr(true)
cfg.Report = boolPtr(DefaultAuditaReport)
}
}
@@ -175,8 +293,11 @@ func applyScriptoriumDefaults(cfg *ScriptoriumConfig) {
if cfg == nil {
return
}
if cfg.Binary == "" {
cfg.Binary = DefaultScriptoriumBinary
}
if cfg.Timeout == "" {
cfg.Timeout = "10m"
cfg.Timeout = DefaultScriptoriumTimeout
}
}
@@ -185,10 +306,10 @@ func applyTrimDefaults(cfg *TrimConfig) {
return
}
if cfg.Bounds.Timeout == "" {
cfg.Bounds.Timeout = "10m"
cfg.Bounds.Timeout = DefaultTrimBoundsTimeout
}
if cfg.Seriatim.Report == nil {
cfg.Seriatim.Report = boolPtr(false)
cfg.Seriatim.Report = boolPtr(DefaultTrimSeriatimReport)
}
}
@@ -197,13 +318,13 @@ func applyNormalizeDefaults(cfg *NormalizeConfig) {
return
}
if cfg.OutputSchema == "" {
cfg.OutputSchema = defaultNormalizeOutputSchema
cfg.OutputSchema = DefaultNormalizeOutputSchema
}
if cfg.OutputPath == "" && !cfg.outputPathWasSet() {
cfg.OutputPath = defaultNormalizeOutputPath
cfg.OutputPath = DefaultNormalizeOutputPath
}
if cfg.Report == nil {
cfg.Report = boolPtr(true)
cfg.Report = boolPtr(DefaultNormalizeReport)
}
}

View File

@@ -15,6 +15,7 @@ func TestLoadAndValidate(t *testing.T) {
wantLoadErr string
wantValidate string
checkDefault bool
wantRoot string
}{
{
name: "valid minimal config",
@@ -39,6 +40,47 @@ inputs:
glossary_file: ./glossary.yml
`,
checkDefault: true,
wantRoot: "/tmp/narratio",
},
{
name: "seriatim and audita sections can be omitted",
pipelineYAML: `workspace:
root: /tmp/narratio
whisperx:
transcribe_url: https://transcription.ai.rakestrawhome.com/transcribe
analyzer:
timeout: 20m
notification:
timeout: 15s
`,
sessionYAML: `session_id: 2026-05-03
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
checkDefault: true,
wantRoot: "/tmp/narratio",
},
{
name: "workspace root defaults when omitted",
pipelineYAML: `whisperx:
transcribe_url: https://transcription.ai.rakestrawhome.com/transcribe
analyzer:
timeout: 20m
notification:
timeout: 15s
`,
sessionYAML: `session_id: 2026-05-03
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
checkDefault: true,
wantRoot: DefaultWorkspaceRoot,
},
{
name: "unknown pipeline field fails",
@@ -72,6 +114,46 @@ inputs:
`,
wantLoadErr: "strict decode failed",
},
{
name: "unknown secrets field fails",
pipelineYAML: `workspace:
root: /tmp/narratio
whisperx:
transcribe_url: https://transcription.ai.rakestrawhome.com/transcribe
seriatim:
binary: seriatim
secrets:
bogus: true
`,
sessionYAML: `session_id: 2026-05-03
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
wantLoadErr: "strict decode failed",
},
{
name: "empty secrets env_dir fails",
pipelineYAML: `workspace:
root: /tmp/narratio
whisperx:
transcribe_url: https://transcription.ai.rakestrawhome.com/transcribe
seriatim:
binary: seriatim
secrets:
env_dir: " "
`,
sessionYAML: `session_id: 2026-05-03
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
wantValidate: "pipeline config \"pipeline.yml\" invalid: pipeline.secrets.env_dir must be non-empty when pipeline.secrets is configured",
},
{
name: "unknown session field fails",
pipelineYAML: `workspace:
@@ -256,7 +338,7 @@ inputs:
wantLoadErr: "strict decode failed",
},
{
name: "missing seriatim binary fails",
name: "missing seriatim binary uses default",
pipelineYAML: `workspace:
root: /tmp/narratio
whisperx:
@@ -271,7 +353,6 @@ inputs:
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
wantValidate: "pipeline config \"pipeline.yml\" invalid: pipeline.seriatim.binary is required",
},
{
name: "invalid seriatim timeout fails",
@@ -372,7 +453,7 @@ inputs:
wantLoadErr: "strict decode failed",
},
{
name: "missing audita binary fails",
name: "missing audita binary uses default",
pipelineYAML: `workspace:
root: /tmp/narratio
whisperx:
@@ -389,7 +470,6 @@ inputs:
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
wantValidate: "pipeline config \"pipeline.yml\" invalid: pipeline.audita.binary is required",
},
{
name: "invalid audita timeout fails",
@@ -413,7 +493,7 @@ inputs:
wantValidate: "pipeline config \"pipeline.yml\" invalid: pipeline.audita.timeout must be a valid duration",
},
{
name: "empty audita modules fails",
name: "empty audita modules is valid override",
pipelineYAML: `workspace:
root: /tmp/narratio
whisperx:
@@ -431,7 +511,6 @@ inputs:
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
wantValidate: "pipeline config \"pipeline.yml\" invalid: pipeline.audita.modules must include at least one module",
},
{
name: "empty audita module item fails",
@@ -501,7 +580,7 @@ inputs:
wantValidate: "pipeline config \"pipeline.yml\" invalid: pipeline.audita.base_url must be a valid URL",
},
{
name: "invalid audita llm_concurrency fails",
name: "legacy audita llm_concurrency field fails strict decode",
pipelineYAML: `workspace:
root: /tmp/narratio
whisperx:
@@ -510,7 +589,7 @@ seriatim:
binary: seriatim
audita:
binary: audita
llm_concurrency: 0
llm_concurrency: 1
`,
sessionYAML: `session_id: 2026-05-03
inputs:
@@ -519,7 +598,49 @@ inputs:
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
wantValidate: "pipeline config \"pipeline.yml\" invalid: pipeline.audita.llm_concurrency must be > 0",
wantLoadErr: "strict decode failed",
},
{
name: "invalid audita total_llm_concurrency fails",
pipelineYAML: `workspace:
root: /tmp/narratio
whisperx:
transcribe_url: https://transcription.ai.rakestrawhome.com/transcribe
seriatim:
binary: seriatim
audita:
binary: audita
total_llm_concurrency: 0
`,
sessionYAML: `session_id: 2026-05-03
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
wantValidate: "pipeline config \"pipeline.yml\" invalid: pipeline.audita.total_llm_concurrency must be > 0",
},
{
name: "invalid audita proposal_llm_concurrency fails",
pipelineYAML: `workspace:
root: /tmp/narratio
whisperx:
transcribe_url: https://transcription.ai.rakestrawhome.com/transcribe
seriatim:
binary: seriatim
audita:
binary: audita
proposal_llm_concurrency: 0
`,
sessionYAML: `session_id: 2026-05-03
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
wantValidate: "pipeline config \"pipeline.yml\" invalid: pipeline.audita.proposal_llm_concurrency must be > 0",
},
{
name: "invalid audita validation_llm_concurrency fails",
@@ -542,6 +663,48 @@ inputs:
`,
wantValidate: "pipeline config \"pipeline.yml\" invalid: pipeline.audita.validation_llm_concurrency must be > 0",
},
{
name: "invalid audita output_schema fails",
pipelineYAML: `workspace:
root: /tmp/narratio
whisperx:
transcribe_url: https://transcription.ai.rakestrawhome.com/transcribe
seriatim:
binary: seriatim
audita:
binary: audita
output_schema: bad
`,
sessionYAML: `session_id: 2026-05-03
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
wantValidate: "pipeline config \"pipeline.yml\" invalid: pipeline.audita.output_schema must be one of: bare-segments, audita-v1",
},
{
name: "invalid audita work_dir_retention fails",
pipelineYAML: `workspace:
root: /tmp/narratio
whisperx:
transcribe_url: https://transcription.ai.rakestrawhome.com/transcribe
seriatim:
binary: seriatim
audita:
binary: audita
work_dir_retention: sometimes
`,
sessionYAML: `session_id: 2026-05-03
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
wantValidate: "pipeline config \"pipeline.yml\" invalid: pipeline.audita.work_dir_retention must be one of: always, auto, never",
},
}
for _, tt := range tests {
@@ -572,6 +735,9 @@ inputs:
t.Fatalf("SessionPath = %q, want %q", cfg.SessionPath, sessionPath)
}
if tt.checkDefault {
if tt.wantRoot != "" && cfg.Pipeline.Workspace.Root != tt.wantRoot {
t.Fatalf("workspace.root = %q, want %q", cfg.Pipeline.Workspace.Root, tt.wantRoot)
}
if cfg.Pipeline.WhisperX.Language != "en" {
t.Fatalf("whisperx.language = %q, want %q", cfg.Pipeline.WhisperX.Language, "en")
}
@@ -590,6 +756,9 @@ inputs:
if cfg.Pipeline.Seriatim.Timeout != "10m" {
t.Fatalf("seriatim.timeout = %q, want %q", cfg.Pipeline.Seriatim.Timeout, "10m")
}
if cfg.Pipeline.Seriatim.Binary != "seriatim" {
t.Fatalf("seriatim.binary = %q, want %q", cfg.Pipeline.Seriatim.Binary, "seriatim")
}
if cfg.Pipeline.Seriatim.OutputSchema != "seriatim-intermediate" {
t.Fatalf("seriatim.output_schema = %q, want %q", cfg.Pipeline.Seriatim.OutputSchema, "seriatim-intermediate")
}
@@ -602,26 +771,32 @@ inputs:
if cfg.Pipeline.Audita.Timeout != "3h" {
t.Fatalf("audita.timeout = %q, want %q", cfg.Pipeline.Audita.Timeout, "3h")
}
if cfg.Pipeline.Audita.Binary != "audita" {
t.Fatalf("audita.binary = %q, want %q", cfg.Pipeline.Audita.Binary, "audita")
}
if cfg.Pipeline.Audita.LLMAPIKeyEnv != "" {
t.Fatalf("audita.llm_api_key_env = %q, want empty by default", cfg.Pipeline.Audita.LLMAPIKeyEnv)
}
if got := strings.Join(cfg.Pipeline.Audita.Modules, ","); got != "glossary,homophones,glossary,spoken_word,grammar,homophones,glossary" {
t.Fatalf("audita.modules = %q, want default sequence", got)
if cfg.Pipeline.Audita.Modules != nil {
t.Fatalf("audita.modules = %#v, want nil default (optional override)", cfg.Pipeline.Audita.Modules)
}
if cfg.Pipeline.Audita.BaseURL != "https://openrouter.ai/api/v1" {
t.Fatalf("audita.base_url = %q, want %q", cfg.Pipeline.Audita.BaseURL, "https://openrouter.ai/api/v1")
if cfg.Pipeline.Audita.BaseURL != "" {
t.Fatalf("audita.base_url = %q, want empty default", cfg.Pipeline.Audita.BaseURL)
}
if cfg.Pipeline.Audita.Model != "openrouter/google/gemma-4-31b-it" {
t.Fatalf("audita.model = %q, want %q", cfg.Pipeline.Audita.Model, "openrouter/google/gemma-4-31b-it")
}
if cfg.Pipeline.Audita.LLMConcurrency == nil || *cfg.Pipeline.Audita.LLMConcurrency != 1 {
t.Fatalf("audita.llm_concurrency = %v, want 1", cfg.Pipeline.Audita.LLMConcurrency)
if cfg.Pipeline.Audita.Model != "" {
t.Fatalf("audita.model = %q, want empty default", cfg.Pipeline.Audita.Model)
}
if cfg.Pipeline.Audita.ValidationModel != "" {
t.Fatalf("audita.validation_model = %q, want empty default", cfg.Pipeline.Audita.ValidationModel)
}
if cfg.Pipeline.Audita.ValidationLLMConcurrency == nil || *cfg.Pipeline.Audita.ValidationLLMConcurrency != 1 {
t.Fatalf("audita.validation_llm_concurrency = %v, want 1", cfg.Pipeline.Audita.ValidationLLMConcurrency)
if cfg.Pipeline.Audita.TotalLLMConcurrency != nil {
t.Fatalf("audita.total_llm_concurrency = %v, want nil default", cfg.Pipeline.Audita.TotalLLMConcurrency)
}
if cfg.Pipeline.Audita.ProposalLLMConcurrency != nil {
t.Fatalf("audita.proposal_llm_concurrency = %v, want nil default", cfg.Pipeline.Audita.ProposalLLMConcurrency)
}
if cfg.Pipeline.Audita.ValidationLLMConcurrency != nil {
t.Fatalf("audita.validation_llm_concurrency = %v, want nil default", cfg.Pipeline.Audita.ValidationLLMConcurrency)
}
if cfg.Pipeline.Audita.Report == nil || *cfg.Pipeline.Audita.Report != true {
t.Fatalf("audita.report = %v, want true", cfg.Pipeline.Audita.Report)
@@ -685,7 +860,8 @@ func TestValidateMissingAudioSource(t *testing.T) {
Modules: []string{"glossary", "homophones"},
BaseURL: "https://openrouter.ai/api/v1",
Model: "openrouter/google/gemma-4-31b-it",
LLMConcurrency: intPtr(1),
TotalLLMConcurrency: intPtr(1),
ProposalLLMConcurrency: intPtr(1),
ValidationModel: "",
ValidationLLMConcurrency: intPtr(1),
Report: boolPtr(true),
@@ -693,6 +869,7 @@ func TestValidateMissingAudioSource(t *testing.T) {
},
Session: &SessionConfig{
SessionID: "2026-05-03",
Campaign: "sample-campaign",
Inputs: SessionInputsConfig{
SpeakersFile: "speakers.yml",
AutocorrectFile: "autocorrect.yml",
@@ -705,7 +882,7 @@ func TestValidateMissingAudioSource(t *testing.T) {
if err == nil {
t.Fatal("expected validation error, got nil")
}
if !strings.Contains(err.Error(), "audio_dir or at least one audio_files") {
if !strings.Contains(err.Error(), "audio_dir, at least one audio_files entry, or audio_s3") {
t.Fatalf("error = %q, want audio source guidance", err.Error())
}
if !strings.Contains(err.Error(), "session config") {
@@ -714,15 +891,59 @@ func TestValidateMissingAudioSource(t *testing.T) {
}
func TestExamplesLoadAndValidate(t *testing.T) {
pipelinePath := filepath.Join("..", "..", "examples", "pipeline.minimal.yml")
sessionPath := filepath.Join("..", "..", "examples", "session.minimal.yml")
cfg, err := Load(pipelinePath, sessionPath)
if err != nil {
t.Fatalf("Load(examples) error = %v", err)
examplesDir := filepath.Join("..", "..", "examples")
tests := []struct {
name string
pipelineFile string
sessionFile string
sessionOpts SessionLoadOptions
}{
{
name: "minimal pipeline with local audio session",
pipelineFile: "pipeline.minimal.yml",
sessionFile: "session.local-audio.yml",
},
{
name: "production pipeline with s3 audio session",
pipelineFile: "pipeline.production.yml",
sessionFile: "session.s3-audio.yml",
},
{
name: "full annotated pipeline with local audio session",
pipelineFile: "pipeline.full.annotated.yml",
sessionFile: "session.local-audio.yml",
},
{
name: "template session renders with session_id option",
pipelineFile: "pipeline.minimal.yml",
sessionFile: "session.template.yml",
sessionOpts: SessionLoadOptions{
SessionID: "2026-05-03",
},
},
}
if err := Validate(cfg); err != nil {
t.Fatalf("Validate(examples) error = %v", err)
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
pipelinePath := filepath.Join(examplesDir, tt.pipelineFile)
sessionPath := filepath.Join(examplesDir, tt.sessionFile)
var (
cfg *Config
err error
)
if strings.TrimSpace(tt.sessionOpts.SessionID) == "" {
cfg, err = Load(pipelinePath, sessionPath)
} else {
cfg, err = LoadWithSessionOptions(pipelinePath, sessionPath, tt.sessionOpts)
}
if err != nil {
t.Fatalf("load example config error = %v", err)
}
if err := Validate(cfg); err != nil {
t.Fatalf("validate example config error = %v", err)
}
})
}
}
@@ -734,6 +955,12 @@ func writeConfigFiles(t *testing.T, pipelineYAML, sessionYAML string) (string, s
}
pipelineYAML += "audita:\n binary: audita\n"
}
if !strings.Contains(sessionYAML, "\ncampaign:") && !strings.HasPrefix(sessionYAML, "campaign:") {
if !strings.HasSuffix(sessionYAML, "\n") {
sessionYAML += "\n"
}
sessionYAML += "campaign: sample-campaign\n"
}
dir := t.TempDir()
pipelinePath := filepath.Join(dir, "pipeline.yml")

View File

@@ -6,11 +6,6 @@ import (
"gopkg.in/yaml.v3"
)
const (
defaultNormalizeOutputPath = "transcripts/normalized.json"
defaultNormalizeOutputSchema = "seriatim-intermediate"
)
// UnmarshalYAML tracks explicit normalize.output_path presence so validation can
// distinguish omitted vs explicitly empty values.
func (cfg *NormalizeConfig) UnmarshalYAML(node *yaml.Node) error {

View File

@@ -24,7 +24,7 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
output_path: artifacts/session_recap.md
inputs:
transcript:
source: processed_transcript
source: narratio.transcript.polished
required: true
vars:
session_id: true
@@ -49,11 +49,19 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
wantLoadErr: "strict decode failed",
},
{
name: "missing binary fails when section present",
name: "missing binary defaults when section present",
scriptoriumYAML: `scriptorium:
timeout: 10m
`,
wantValidateErr: "pipeline.scriptorium.binary is required",
assert: func(t *testing.T, cfg *Config) {
t.Helper()
if cfg.Pipeline.Scriptorium == nil {
t.Fatal("scriptorium config should be present")
}
if cfg.Pipeline.Scriptorium.Binary != "scriptorium" {
t.Fatalf("scriptorium.binary = %q, want scriptorium", cfg.Pipeline.Scriptorium.Binary)
}
},
},
{
name: "enabled artifact missing prompt id fails",
@@ -96,7 +104,7 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
output_path: artifacts/session_recap.md
inputs:
transcript:
source: processed_transcript
source: narratio.transcript.polished
required: true
previous_recap:
source: previous_session_artifact
@@ -108,6 +116,37 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
output_kind: session_recap
`,
},
{
name: "canonical artifact source is accepted",
scriptoriumYAML: `scriptorium:
binary: scriptorium
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
output_path: artifacts/session_recap.md
inputs:
transcript:
source: narratio.transcript.trimmed
required: true
`,
},
{
name: "unknown artifact source fails validation",
scriptoriumYAML: `scriptorium:
binary: scriptorium
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
output_path: artifacts/session_recap.md
inputs:
transcript:
source: narratio.unknown
required: true
`,
wantValidateErr: `pipeline.scriptorium.artifacts.session_recap.inputs.transcript.source "narratio.unknown" is unsupported`,
},
{
name: "artifact render_debug override is accepted",
scriptoriumYAML: `scriptorium:
@@ -121,7 +160,7 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
output_path: artifacts/session_recap.md
inputs:
transcript:
source: processed_transcript
source: narratio.transcript.polished
required: true
`,
assert: func(t *testing.T, cfg *Config) {
@@ -143,7 +182,7 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
output_path: artifacts/session_recap.md
inputs:
transcript:
source: processed_transcript
source: narratio.transcript.polished
required: true
player_summary:
enabled: true
@@ -153,7 +192,7 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
timeout: 3m
inputs:
transcript:
source: processed_transcript
source: narratio.transcript.polished
required: true
`,
assert: func(t *testing.T, cfg *Config) {
@@ -166,6 +205,195 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
}
},
},
{
name: "valid artifact dependency is accepted",
scriptoriumYAML: `scriptorium:
binary: scriptorium
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
output_path: artifacts/session_recap.md
inputs:
transcript:
source: narratio.transcript.trimmed
required: true
player_handout:
enabled: true
depends_on:
- session_recap
prompt_id: dnd.player_handout
output_path: artifacts/player_handout.md
inputs:
recap:
source: narratio.artifact.session_recap
required: true
`,
},
{
name: "valid dependency on disabled artifact with output path is accepted",
scriptoriumYAML: `scriptorium:
binary: scriptorium
artifacts:
session_recap:
enabled: false
output_path: artifacts/session_recap.md
player_handout:
enabled: true
depends_on:
- session_recap
prompt_id: dnd.player_handout
output_path: artifacts/player_handout.md
inputs:
recap:
source: narratio.artifact.session_recap
required: true
`,
},
{
name: "invalid artifact name fails validation",
scriptoriumYAML: `scriptorium:
binary: scriptorium
artifacts:
SessionRecap:
enabled: true
prompt_id: dnd.session_recap
output_path: artifacts/session_recap.md
`,
wantValidateErr: "pipeline.scriptorium.artifacts keys must match ^[a-z][a-z0-9_]*$",
},
{
name: "artifact output path outside artifacts root fails validation",
scriptoriumYAML: `scriptorium:
binary: scriptorium
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
output_path: transcripts/session_recap.md
`,
wantValidateErr: "pipeline.scriptorium.artifacts.session_recap.output_path must be under artifacts/",
},
{
name: "missing depends_on for artifact source fails validation",
scriptoriumYAML: `scriptorium:
binary: scriptorium
artifacts:
session_recap:
enabled: true
prompt_id: dnd.session_recap
output_path: artifacts/session_recap.md
inputs:
transcript:
source: narratio.transcript.polished
required: true
player_handout:
enabled: true
prompt_id: dnd.player_handout
output_path: artifacts/player_handout.md
inputs:
recap:
source: narratio.artifact.session_recap
required: true
`,
wantValidateErr: `pipeline.scriptorium.artifacts.player_handout.inputs.recap.source "narratio.artifact.session_recap" requires depends_on entry "session_recap"`,
},
{
name: "dependency on unknown artifact fails validation",
scriptoriumYAML: `scriptorium:
binary: scriptorium
artifacts:
player_handout:
enabled: true
depends_on:
- session_recap
prompt_id: dnd.player_handout
output_path: artifacts/player_handout.md
inputs:
recap:
source: narratio.artifact.session_recap
required: true
`,
wantValidateErr: `pipeline.scriptorium.artifacts.player_handout.depends_on[0] "session_recap" is not a configured artifact key`,
},
{
name: "self dependency fails validation",
scriptoriumYAML: `scriptorium:
binary: scriptorium
artifacts:
session_recap:
enabled: true
depends_on:
- session_recap
prompt_id: dnd.session_recap
output_path: artifacts/session_recap.md
`,
wantValidateErr: "pipeline.scriptorium.artifacts.session_recap.depends_on must not include itself",
},
{
name: "enabled dependency cycle fails validation",
scriptoriumYAML: `scriptorium:
binary: scriptorium
artifacts:
artifact_a:
enabled: true
depends_on:
- artifact_b
prompt_id: dnd.a
output_path: artifacts/a.md
inputs:
b:
source: narratio.artifact.artifact_b
required: true
artifact_b:
enabled: true
depends_on:
- artifact_a
prompt_id: dnd.b
output_path: artifacts/b.md
inputs:
a:
source: narratio.artifact.artifact_a
required: true
`,
wantValidateErr: "pipeline.scriptorium.artifacts enabled dependencies must not contain cycles",
},
{
name: "artifact source typo fails validation",
scriptoriumYAML: `scriptorium:
binary: scriptorium
artifacts:
player_handout:
enabled: true
prompt_id: dnd.player_handout
output_path: artifacts/player_handout.md
inputs:
recap:
source: narratio.artifact.session-recap
required: true
`,
wantValidateErr: `pipeline.scriptorium.artifacts.player_handout.inputs.recap.source "narratio.artifact.session-recap" is unsupported`,
},
{
name: "referenced disabled artifact missing output path fails validation",
scriptoriumYAML: `scriptorium:
binary: scriptorium
artifacts:
session_recap:
enabled: false
player_handout:
enabled: true
depends_on:
- session_recap
prompt_id: dnd.player_handout
output_path: artifacts/player_handout.md
inputs:
recap:
source: narratio.artifact.session_recap
required: true
`,
wantValidateErr: "pipeline.scriptorium.artifacts.session_recap.output_path is required when artifact is referenced",
},
}
for _, tt := range tests {
@@ -219,6 +447,7 @@ audita:
`
const testSessionBaseYAML = `session_id: 2026-05-03
campaign: test-campaign
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml

View File

@@ -0,0 +1,156 @@
package config
import (
"os"
"path/filepath"
"strings"
"testing"
)
func TestLoadSessionWithOptionsRendersCompactPlaceholder(t *testing.T) {
dir := t.TempDir()
sessionPath := filepath.Join(dir, "session.yml")
sessionYAML := `session_id: "{{session_id}}"
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
if err := os.WriteFile(sessionPath, []byte(sessionYAML), 0o644); err != nil {
t.Fatalf("write session.yml: %v", err)
}
cfg, err := LoadSessionWithOptions(sessionPath, SessionLoadOptions{SessionID: "2026-04-04"})
if err != nil {
t.Fatalf("LoadSessionWithOptions() error = %v", err)
}
if cfg.SessionID != "2026-04-04" {
t.Fatalf("SessionID = %q, want 2026-04-04", cfg.SessionID)
}
}
func TestLoadSessionWithOptionsRendersSpacedPlaceholder(t *testing.T) {
dir := t.TempDir()
sessionPath := filepath.Join(dir, "session.yml")
sessionYAML := `session_id: "{{ session_id }}"
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
if err := os.WriteFile(sessionPath, []byte(sessionYAML), 0o644); err != nil {
t.Fatalf("write session.yml: %v", err)
}
cfg, err := LoadSessionWithOptions(sessionPath, SessionLoadOptions{SessionID: "2026-04-04"})
if err != nil {
t.Fatalf("LoadSessionWithOptions() error = %v", err)
}
if cfg.SessionID != "2026-04-04" {
t.Fatalf("SessionID = %q, want 2026-04-04", cfg.SessionID)
}
}
func TestLoadSessionWithOptionsUnresolvedPlaceholderFails(t *testing.T) {
dir := t.TempDir()
sessionPath := filepath.Join(dir, "session.yml")
sessionYAML := `session_id: "{{ session_id }}"
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
if err := os.WriteFile(sessionPath, []byte(sessionYAML), 0o644); err != nil {
t.Fatalf("write session.yml: %v", err)
}
_, err := LoadSessionWithOptions(sessionPath, SessionLoadOptions{})
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "unresolved template variable") {
t.Fatalf("error = %q, want unresolved-variable context", err.Error())
}
if !strings.Contains(err.Error(), "session_id") {
t.Fatalf("error = %q, want session_id variable", err.Error())
}
}
func TestLoadSessionWithOptionsMismatchFails(t *testing.T) {
dir := t.TempDir()
sessionPath := filepath.Join(dir, "session.yml")
sessionYAML := `session_id: 2026-05-03
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
if err := os.WriteFile(sessionPath, []byte(sessionYAML), 0o644); err != nil {
t.Fatalf("write session.yml: %v", err)
}
_, err := LoadSessionWithOptions(sessionPath, SessionLoadOptions{SessionID: "2026-04-04"})
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "session_id mismatch") {
t.Fatalf("error = %q, want mismatch context", err.Error())
}
}
func TestLoadSessionWithOptionsUnknownFieldStillRejectedAfterRendering(t *testing.T) {
dir := t.TempDir()
sessionPath := filepath.Join(dir, "session.yml")
sessionYAML := `session_id: "{{ session_id }}"
campaign: sample-campaign
unknown_field: true
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
if err := os.WriteFile(sessionPath, []byte(sessionYAML), 0o644); err != nil {
t.Fatalf("write session.yml: %v", err)
}
_, err := LoadSessionWithOptions(sessionPath, SessionLoadOptions{SessionID: "2026-04-04"})
if err == nil {
t.Fatal("expected error, got nil")
}
if !strings.Contains(err.Error(), "strict decode failed") {
t.Fatalf("error = %q, want strict-decode context", err.Error())
}
}
func TestLoadSessionWithOptionsConcreteSessionStillLoads(t *testing.T) {
dir := t.TempDir()
sessionPath := filepath.Join(dir, "session.yml")
sessionYAML := `session_id: 2026-05-03
campaign: sample-campaign
inputs:
audio_dir: ./audio
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
if err := os.WriteFile(sessionPath, []byte(sessionYAML), 0o644); err != nil {
t.Fatalf("write session.yml: %v", err)
}
cfg, err := LoadSessionWithOptions(sessionPath, SessionLoadOptions{})
if err != nil {
t.Fatalf("LoadSessionWithOptions() error = %v", err)
}
if cfg.SessionID != "2026-05-03" {
t.Fatalf("SessionID = %q, want 2026-05-03", cfg.SessionID)
}
}

View File

@@ -0,0 +1,315 @@
package config
import (
"strings"
"testing"
)
func TestStorageS3DefaultsAndValidation(t *testing.T) {
pipelineYAML := testPipelineBaseYAML + `
storage:
backend: s3
s3:
bucket: my-dnd-archive
`
pipelinePath, sessionPath := writeConfigFiles(t, pipelineYAML, testSessionBaseYAML)
cfg, err := Load(pipelinePath, sessionPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
if cfg.Pipeline.Storage.S3 == nil {
t.Fatal("storage.s3 should be initialized")
}
if cfg.Pipeline.Storage.S3.RootPrefix != "dnd" {
t.Fatalf("storage.s3.root_prefix = %q, want dnd", cfg.Pipeline.Storage.S3.RootPrefix)
}
if cfg.Pipeline.Storage.S3.AccessKeyIDEnv != DefaultS3AccessKeyIDEnv {
t.Fatalf("storage.s3.access_key_id_env = %q, want %q", cfg.Pipeline.Storage.S3.AccessKeyIDEnv, DefaultS3AccessKeyIDEnv)
}
if cfg.Pipeline.Storage.S3.SecretKeyEnv != DefaultS3SecretAccessKeyEnv {
t.Fatalf("storage.s3.secret_access_key_env = %q, want %q", cfg.Pipeline.Storage.S3.SecretKeyEnv, DefaultS3SecretAccessKeyEnv)
}
if cfg.Pipeline.Storage.S3.ForcePathStyle {
t.Fatalf("storage.s3.force_path_style = true, want false default")
}
if err := Validate(cfg); err != nil {
t.Fatalf("Validate() error = %v", err)
}
}
func TestStorageS3CredentialEnvNamesLoadAndValidate(t *testing.T) {
pipelineYAML := testPipelineBaseYAML + `
storage:
backend: s3
s3:
bucket: my-dnd-archive
access_key_id_env: CUSTOM_KEY_ID
secret_access_key_env: CUSTOM_SECRET
`
pipelinePath, sessionPath := writeConfigFiles(t, pipelineYAML, testSessionBaseYAML)
cfg, err := Load(pipelinePath, sessionPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
if cfg.Pipeline.Storage.S3.AccessKeyIDEnv != "CUSTOM_KEY_ID" {
t.Fatalf("storage.s3.access_key_id_env = %q, want CUSTOM_KEY_ID", cfg.Pipeline.Storage.S3.AccessKeyIDEnv)
}
if cfg.Pipeline.Storage.S3.SecretKeyEnv != "CUSTOM_SECRET" {
t.Fatalf("storage.s3.secret_access_key_env = %q, want CUSTOM_SECRET", cfg.Pipeline.Storage.S3.SecretKeyEnv)
}
if err := Validate(cfg); err != nil {
t.Fatalf("Validate() error = %v", err)
}
}
func TestStorageS3CredentialEnvValidation(t *testing.T) {
tests := []struct {
name string
pipelineYML string
wantErr string
}{
{
name: "invalid access key env name",
pipelineYML: testPipelineBaseYAML + `
storage:
backend: s3
s3:
bucket: my-dnd-archive
access_key_id_env: "123BAD"
`,
wantErr: "pipeline.storage.s3.access_key_id_env must be a valid environment variable name",
},
{
name: "invalid secret key env name",
pipelineYML: testPipelineBaseYAML + `
storage:
backend: s3
s3:
bucket: my-dnd-archive
secret_access_key_env: "bad-name"
`,
wantErr: "pipeline.storage.s3.secret_access_key_env must be a valid environment variable name",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
pipelinePath, sessionPath := writeConfigFiles(t, tt.pipelineYML, testSessionBaseYAML)
cfg, err := Load(pipelinePath, sessionPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
err = Validate(cfg)
if err == nil || !strings.Contains(err.Error(), tt.wantErr) {
t.Fatalf("Validate() error = %v, want to contain %q", err, tt.wantErr)
}
})
}
}
func TestSpoolAndArchiveDefaults(t *testing.T) {
pipelinePath, sessionPath := writeConfigFiles(t, testPipelineBaseYAML, testSessionBaseYAML)
cfg, err := Load(pipelinePath, sessionPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
if cfg.Pipeline.Spool.Root != "/var/spool/narratio" {
t.Fatalf("spool.root = %q, want /var/spool/narratio", cfg.Pipeline.Spool.Root)
}
if cfg.Pipeline.Spool.DeleteAudioAfterArchive {
t.Fatalf("spool.delete_audio_after_archive = true, want false")
}
if cfg.Pipeline.Workspace.CleanupAfterArchive {
t.Fatalf("workspace.cleanup_after_archive = true, want false")
}
if cfg.Pipeline.Archive == nil {
t.Fatal("archive should be initialized by defaults")
}
if cfg.Pipeline.Archive.Enabled == nil || !*cfg.Pipeline.Archive.Enabled {
t.Fatalf("archive.enabled = %#v, want true", cfg.Pipeline.Archive.Enabled)
}
if cfg.Pipeline.Archive.UploadRun == nil || !*cfg.Pipeline.Archive.UploadRun {
t.Fatalf("archive.upload_run = %#v, want true", cfg.Pipeline.Archive.UploadRun)
}
if len(cfg.Pipeline.Archive.PromoteArtifacts) != 2 {
t.Fatalf("archive.promote_artifacts len = %d, want 2 defaults", len(cfg.Pipeline.Archive.PromoteArtifacts))
}
for i, item := range cfg.Pipeline.Archive.PromoteArtifacts {
if item.Required == nil || !*item.Required {
t.Fatalf("archive.promote_artifacts[%d].required = %#v, want true", i, item.Required)
}
}
}
func TestArchivePromotionPathValidation(t *testing.T) {
tests := []struct {
name string
ruleYML string
wantErr string
}{
{
name: "absolute from path rejected",
ruleYML: `archive:
promote_artifacts:
- from: "/transcripts/trimmed.json"
to: "transcripts/trimmed.json"
`,
wantErr: "must be a relative path",
},
{
name: "traversal to path rejected",
ruleYML: `archive:
promote_artifacts:
- from: "transcripts/trimmed.json"
to: "../trimmed.json"
`,
wantErr: "must not contain path traversal",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
pipelineYAML := testPipelineBaseYAML + "\n" + tt.ruleYML
pipelinePath, sessionPath := writeConfigFiles(t, pipelineYAML, testSessionBaseYAML)
cfg, err := Load(pipelinePath, sessionPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
err = Validate(cfg)
if err == nil || !strings.Contains(err.Error(), tt.wantErr) {
t.Fatalf("Validate() error = %v, want to contain %q", err, tt.wantErr)
}
})
}
}
func TestSessionAudioS3Validation(t *testing.T) {
tests := []struct {
name string
sessionYAML string
wantErr string
}{
{
name: "valid audio_s3 prefix",
sessionYAML: `session_id: 2026-05-03
campaign: forsaken
inputs:
audio_s3:
prefix: audio/
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
},
{
name: "invalid audio_s3 absolute prefix",
sessionYAML: `session_id: 2026-05-03
campaign: forsaken
inputs:
audio_s3:
prefix: /audio/
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
wantErr: "session.inputs.audio_s3.prefix must be a relative path",
},
{
name: "invalid audio_s3 traversal prefix",
sessionYAML: `session_id: 2026-05-03
campaign: forsaken
inputs:
audio_s3:
prefix: ../audio/
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
wantErr: "session.inputs.audio_s3.prefix must not contain path traversal",
},
{
name: "local and s3 audio conflict",
sessionYAML: `session_id: 2026-05-03
campaign: forsaken
inputs:
audio_dir: ./audio
audio_s3:
prefix: audio/
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`,
wantErr: "mutually exclusive",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
pipelineYAML := testPipelineBaseYAML + `
storage:
backend: s3
s3:
bucket: my-dnd-archive
`
pipelinePath, sessionPath := writeConfigFiles(t, pipelineYAML, tt.sessionYAML)
cfg, err := Load(pipelinePath, sessionPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
err = Validate(cfg)
if tt.wantErr != "" {
if err == nil || !strings.Contains(err.Error(), tt.wantErr) {
t.Fatalf("Validate() error = %v, want to contain %q", err, tt.wantErr)
}
return
}
if err != nil {
t.Fatalf("Validate() error = %v", err)
}
})
}
}
func TestStorageS3BucketRequiredWhenS3DependentFeatureEnabled(t *testing.T) {
pipelineYAML := testPipelineBaseYAML + `
storage:
backend: s3
`
sessionYAML := `session_id: 2026-05-03
campaign: forsaken
inputs:
audio_s3:
prefix: audio/
speakers_file: ./speakers.yml
autocorrect_file: ./autocorrect.yml
glossary_file: ./glossary.yml
`
pipelinePath, sessionPath := writeConfigFiles(t, pipelineYAML, sessionYAML)
cfg, err := Load(pipelinePath, sessionPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
err = Validate(cfg)
if err == nil || !strings.Contains(err.Error(), "pipeline.storage.s3.bucket is required") {
t.Fatalf("Validate() error = %v, want bucket requirement", err)
}
}
func TestLocalAudioConfigStillValid(t *testing.T) {
pipelinePath, sessionPath := writeConfigFiles(t, testPipelineBaseYAML, testSessionBaseYAML)
cfg, err := Load(pipelinePath, sessionPath)
if err != nil {
t.Fatalf("Load() error = %v", err)
}
if err := Validate(cfg); err != nil {
t.Fatalf("Validate() error = %v", err)
}
}

View File

@@ -3,6 +3,8 @@ package config
import (
"fmt"
"net/url"
"path/filepath"
"regexp"
"strings"
"time"
)
@@ -25,6 +27,9 @@ func Validate(cfg *Config) error {
if err := validateSession(cfg.Session); err != nil {
return fmt.Errorf("session config %q invalid: %w", shortName(cfg.SessionPath, "session.yml"), err)
}
if err := validateCrossConfig(cfg.Pipeline, cfg.Session); err != nil {
return fmt.Errorf("pipeline/session config invalid: %w", err)
}
return nil
}
@@ -33,6 +38,18 @@ func validatePipeline(cfg *PipelineConfig) error {
if strings.TrimSpace(cfg.Workspace.Root) == "" {
return fmt.Errorf("pipeline.workspace.root is required")
}
if err := validateSecrets(cfg.Secrets); err != nil {
return err
}
if err := validateStorage(cfg.Storage); err != nil {
return err
}
if err := validateSpool(cfg.Spool); err != nil {
return err
}
if err := validateArchive(cfg.Archive); err != nil {
return err
}
if err := validateWhisperX(cfg.WhisperX); err != nil {
return err
}
@@ -61,6 +78,64 @@ func validatePipeline(cfg *PipelineConfig) error {
return nil
}
func validateStorage(cfg StorageConfig) error {
if cfg.S3 == nil {
return nil
}
if strings.TrimSpace(cfg.S3.RootPrefix) == "" {
return fmt.Errorf("pipeline.storage.s3.root_prefix must be non-empty")
}
if err := validateRelativeSafePath("pipeline.storage.s3.root_prefix", cfg.S3.RootPrefix); err != nil {
return err
}
if cfg.S3.Endpoint != "" && strings.TrimSpace(cfg.S3.Endpoint) == "" {
return fmt.Errorf("pipeline.storage.s3.endpoint must be non-empty when provided")
}
if err := validateEnvVarNameField("pipeline.storage.s3.access_key_id_env", cfg.S3.AccessKeyIDEnv); err != nil {
return err
}
if err := validateEnvVarNameField("pipeline.storage.s3.secret_access_key_env", cfg.S3.SecretKeyEnv); err != nil {
return err
}
return nil
}
func validateSpool(cfg SpoolConfig) error {
return nil
}
func validateArchive(cfg *ArchiveConfig) error {
if cfg == nil {
return nil
}
for i, item := range cfg.PromoteArtifacts {
prefix := fmt.Sprintf("pipeline.archive.promote_artifacts[%d]", i)
if strings.TrimSpace(item.From) == "" {
return fmt.Errorf("%s.from is required", prefix)
}
if strings.TrimSpace(item.To) == "" {
return fmt.Errorf("%s.to is required", prefix)
}
if err := validateRelativeSafePath(prefix+".from", item.From); err != nil {
return err
}
if err := validateRelativeSafePath(prefix+".to", item.To); err != nil {
return err
}
}
return nil
}
func validateSecrets(cfg *SecretsConfig) error {
if cfg == nil {
return nil
}
if strings.TrimSpace(cfg.EnvDir) == "" {
return fmt.Errorf("pipeline.secrets.env_dir must be non-empty when pipeline.secrets is configured")
}
return nil
}
func validateNormalize(cfg *NormalizeConfig) error {
if cfg == nil {
return nil
@@ -187,15 +262,12 @@ func validateAudita(cfg AuditaConfig) error {
if err := validateDuration("pipeline.audita.timeout", cfg.Timeout); err != nil {
return err
}
if len(cfg.Modules) == 0 {
return fmt.Errorf("pipeline.audita.modules must include at least one module")
}
for i, mod := range cfg.Modules {
m := strings.TrimSpace(mod)
if m == "" {
for i, m := range cfg.Modules {
module := strings.TrimSpace(m)
if module == "" {
return fmt.Errorf("pipeline.audita.modules[%d] must be non-empty", i)
}
switch m {
switch module {
case "glossary", "homophones", "spoken_word", "grammar":
default:
return fmt.Errorf("pipeline.audita.modules[%d] must be one of: glossary, homophones, spoken_word, grammar", i)
@@ -210,21 +282,31 @@ func validateAudita(cfg AuditaConfig) error {
return fmt.Errorf("pipeline.audita.base_url must be a valid URL")
}
}
if strings.TrimSpace(cfg.Model) == "" {
return fmt.Errorf("pipeline.audita.model is required")
if cfg.TotalLLMConcurrency != nil && *cfg.TotalLLMConcurrency <= 0 {
return fmt.Errorf("pipeline.audita.total_llm_concurrency must be > 0")
}
if cfg.LLMConcurrency == nil {
return fmt.Errorf("pipeline.audita.llm_concurrency must be set (defaults should populate this)")
if cfg.ProposalLLMConcurrency != nil && *cfg.ProposalLLMConcurrency <= 0 {
return fmt.Errorf("pipeline.audita.proposal_llm_concurrency must be > 0")
}
if *cfg.LLMConcurrency <= 0 {
return fmt.Errorf("pipeline.audita.llm_concurrency must be > 0")
}
if cfg.ValidationLLMConcurrency == nil {
return fmt.Errorf("pipeline.audita.validation_llm_concurrency must be set (defaults should populate this)")
}
if *cfg.ValidationLLMConcurrency <= 0 {
if cfg.ValidationLLMConcurrency != nil && *cfg.ValidationLLMConcurrency <= 0 {
return fmt.Errorf("pipeline.audita.validation_llm_concurrency must be > 0")
}
if strings.TrimSpace(cfg.TranscriptDescription) == "" && cfg.TranscriptDescription != "" {
return fmt.Errorf("pipeline.audita.transcript_description must be non-empty when provided")
}
if strings.TrimSpace(cfg.ConfigPath) == "" && cfg.ConfigPath != "" {
return fmt.Errorf("pipeline.audita.config_path must be non-empty when provided")
}
switch strings.TrimSpace(cfg.OutputSchema) {
case "", "bare-segments", "audita-v1":
default:
return fmt.Errorf("pipeline.audita.output_schema must be one of: bare-segments, audita-v1")
}
switch strings.TrimSpace(cfg.WorkDirRetention) {
case "", "always", "auto", "never":
default:
return fmt.Errorf("pipeline.audita.work_dir_retention must be one of: always, auto, never")
}
return nil
}
@@ -242,28 +324,78 @@ func validateScriptorium(cfg *ScriptoriumConfig) error {
return err
}
for artifactName, artifactCfg := range cfg.Artifacts {
trimmedArtifactName := strings.TrimSpace(artifactName)
if trimmedArtifactName == "" {
return fmt.Errorf("pipeline.scriptorium.artifacts keys must be non-empty")
configuredArtifacts := make(map[string]struct{}, len(cfg.Artifacts))
referencedArtifacts := make(map[string]struct{})
for artifactName := range cfg.Artifacts {
if !scriptoriumArtifactKeyRE.MatchString(strings.TrimSpace(artifactName)) {
return fmt.Errorf("pipeline.scriptorium.artifacts keys must match ^[a-z][a-z0-9_]*$")
}
configuredArtifacts[artifactName] = struct{}{}
}
for artifactName, artifactCfg := range cfg.Artifacts {
if artifactCfg.Enabled && strings.TrimSpace(artifactCfg.PromptID) == "" {
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.prompt_id is required when enabled", artifactName)
}
if artifactCfg.Enabled && strings.TrimSpace(artifactCfg.OutputPath) == "" {
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.output_path is required when enabled", artifactName)
}
if strings.TrimSpace(artifactCfg.OutputPath) != "" {
pathField := "pipeline.scriptorium.artifacts." + artifactName + ".output_path"
if err := validateRelativeSafePath(pathField, artifactCfg.OutputPath); err != nil {
return err
}
if err := validatePathWithinRoot(pathField, artifactCfg.OutputPath, DefaultScriptoriumArtifactOutputRoot); err != nil {
return err
}
}
if err := validateDuration("pipeline.scriptorium.artifacts."+artifactName+".timeout", artifactCfg.Timeout); err != nil {
return err
}
depSet := make(map[string]struct{}, len(artifactCfg.DependsOn))
for i, depName := range artifactCfg.DependsOn {
trimmedDep := strings.TrimSpace(depName)
field := fmt.Sprintf("pipeline.scriptorium.artifacts.%s.depends_on[%d]", artifactName, i)
if trimmedDep == "" {
return fmt.Errorf("%s must be non-empty", field)
}
if _, ok := configuredArtifacts[trimmedDep]; !ok {
return fmt.Errorf("%s %q is not a configured artifact key", field, depName)
}
if trimmedDep == artifactName {
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.depends_on must not include itself", artifactName)
}
depSet[trimmedDep] = struct{}{}
referencedArtifacts[trimmedDep] = struct{}{}
}
for inputName, inputCfg := range artifactCfg.Inputs {
trimmedInputName := strings.TrimSpace(inputName)
if trimmedInputName == "" {
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.inputs keys must be non-empty", artifactName)
}
if strings.TrimSpace(inputCfg.Source) == "" {
source := strings.TrimSpace(inputCfg.Source)
if source == "" {
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.inputs.%s.source is required", artifactName, inputName)
}
referencedArtifact, err := validateScriptoriumInputSource(artifactName, inputName, source, configuredArtifacts)
if err != nil {
return err
}
if referencedArtifact != "" {
if _, ok := depSet[referencedArtifact]; !ok {
return fmt.Errorf(
"pipeline.scriptorium.artifacts.%s.inputs.%s.source %q requires depends_on entry %q",
artifactName,
inputName,
source,
referencedArtifact,
)
}
referencedArtifacts[referencedArtifact] = struct{}{}
}
}
for varName, varValue := range artifactCfg.Vars {
if strings.TrimSpace(varName) == "" {
@@ -277,6 +409,17 @@ func validateScriptorium(cfg *ScriptoriumConfig) error {
}
}
for artifactName := range referencedArtifacts {
artifactCfg := cfg.Artifacts[artifactName]
if strings.TrimSpace(artifactCfg.OutputPath) == "" {
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.output_path is required when artifact is referenced", artifactName)
}
}
if err := validateEnabledArtifactDependencyCycles(cfg.Artifacts); err != nil {
return err
}
return nil
}
@@ -284,6 +427,9 @@ func validateSession(cfg *SessionConfig) error {
if strings.TrimSpace(cfg.SessionID) == "" {
return fmt.Errorf("session.session_id is required")
}
if strings.TrimSpace(cfg.Campaign) == "" {
return fmt.Errorf("session.campaign is required")
}
if strings.TrimSpace(cfg.Inputs.SpeakersFile) == "" {
return fmt.Errorf("session.inputs.speakers_file is required")
@@ -297,13 +443,205 @@ func validateSession(cfg *SessionConfig) error {
hasAudioDir := strings.TrimSpace(cfg.Inputs.AudioDir) != ""
hasAudioFiles := len(cfg.Inputs.AudioFiles) > 0
if !hasAudioDir && !hasAudioFiles {
return fmt.Errorf("session.inputs requires audio_dir or at least one audio_files entry")
hasAudioS3 := cfg.Inputs.AudioS3 != nil
if hasAudioS3 {
if strings.TrimSpace(cfg.Inputs.AudioS3.Prefix) == "" {
return fmt.Errorf("session.inputs.audio_s3.prefix is required when session.inputs.audio_s3 is configured")
}
if err := validateRelativeSafePath("session.inputs.audio_s3.prefix", cfg.Inputs.AudioS3.Prefix); err != nil {
return err
}
}
if hasAudioS3 && (hasAudioDir || hasAudioFiles) {
return fmt.Errorf("session.inputs.audio_dir/audio_files and session.inputs.audio_s3 are mutually exclusive")
}
if !hasAudioDir && !hasAudioFiles && !hasAudioS3 {
return fmt.Errorf("session.inputs requires audio_dir, at least one audio_files entry, or audio_s3")
}
return nil
}
func validateCrossConfig(pipeline *PipelineConfig, session *SessionConfig) error {
if pipeline == nil || session == nil {
return nil
}
if pipeline.Storage.S3 == nil {
return nil
}
audioS3Enabled := session.Inputs.AudioS3 != nil
archiveUploadEnabled := archiveUploadConfiguredForS3(pipeline)
if (audioS3Enabled || archiveUploadEnabled) && strings.TrimSpace(pipeline.Storage.S3.Bucket) == "" {
return fmt.Errorf("pipeline.storage.s3.bucket is required when S3 session audio or archive upload is enabled")
}
return nil
}
func archiveUploadConfiguredForS3(pipeline *PipelineConfig) bool {
if pipeline == nil || pipeline.Archive == nil {
return false
}
if !strings.EqualFold(strings.TrimSpace(pipeline.Storage.Backend), "s3") {
return false
}
enabled := true
if pipeline.Archive.Enabled != nil {
enabled = *pipeline.Archive.Enabled
}
upload := true
if pipeline.Archive.UploadRun != nil {
upload = *pipeline.Archive.UploadRun
}
return enabled && upload
}
var windowsAbsPathRE = regexp.MustCompile(`^[A-Za-z]:[\\/].*`)
var envVarNameRE = regexp.MustCompile(`^[A-Za-z_][A-Za-z0-9_]*$`)
var scriptoriumArtifactKeyRE = regexp.MustCompile(`^[a-z][a-z0-9_]*$`)
var narratioArtifactSourceRE = regexp.MustCompile(`^narratio\.artifact\.([a-z][a-z0-9_]*)$`)
func validateScriptoriumInputSource(artifactName, inputName, source string, configuredArtifacts map[string]struct{}) (string, error) {
if isStaticSupportedScriptoriumInputSource(source) {
return "", nil
}
matches := narratioArtifactSourceRE.FindStringSubmatch(source)
if len(matches) != 2 {
return "", fmt.Errorf(
"pipeline.scriptorium.artifacts.%s.inputs.%s.source %q is unsupported",
artifactName,
inputName,
source,
)
}
referenced := matches[1]
if _, ok := configuredArtifacts[referenced]; !ok {
return "", fmt.Errorf(
"pipeline.scriptorium.artifacts.%s.inputs.%s.source %q references unknown artifact %q",
artifactName,
inputName,
source,
referenced,
)
}
return referenced, nil
}
func isStaticSupportedScriptoriumInputSource(source string) bool {
switch strings.TrimSpace(source) {
case "previous_session_artifact":
return true
case "narratio.transcript.merged":
return true
case "narratio.transcript.polished":
return true
case "narratio.transcript.full":
return true
case "narratio.transcript.trimmed":
return true
case "narratio.bounds.session":
return true
default:
return false
}
}
func validateEnvVarNameField(fieldName, value string) error {
trimmed := strings.TrimSpace(value)
if trimmed == "" {
return fmt.Errorf("%s must be non-empty", fieldName)
}
if !envVarNameRE.MatchString(trimmed) {
return fmt.Errorf("%s must be a valid environment variable name", fieldName)
}
return nil
}
func validateRelativeSafePath(fieldName, value string) error {
trimmed := strings.TrimSpace(value)
if trimmed == "" {
return fmt.Errorf("%s must be non-empty", fieldName)
}
if filepath.IsAbs(trimmed) || strings.HasPrefix(trimmed, "/") || strings.HasPrefix(trimmed, "\\") || windowsAbsPathRE.MatchString(trimmed) {
return fmt.Errorf("%s must be a relative path", fieldName)
}
normalized := strings.ReplaceAll(trimmed, "\\", "/")
for _, segment := range strings.Split(normalized, "/") {
if segment == ".." {
return fmt.Errorf("%s must not contain path traversal", fieldName)
}
}
return nil
}
func validatePathWithinRoot(fieldName, value, root string) error {
normalizedValue := filepath.ToSlash(filepath.Clean(strings.TrimSpace(value)))
normalizedRoot := filepath.ToSlash(filepath.Clean(strings.TrimSpace(root)))
if normalizedValue == normalizedRoot {
return nil
}
if strings.HasPrefix(normalizedValue, normalizedRoot+"/") {
return nil
}
return fmt.Errorf("%s must be under %s/", fieldName, normalizedRoot)
}
func validateEnabledArtifactDependencyCycles(artifacts map[string]ScriptoriumArtifactConfig) error {
if len(artifacts) == 0 {
return nil
}
enabled := make(map[string]struct{}, len(artifacts))
graph := make(map[string][]string, len(artifacts))
for name, cfg := range artifacts {
if !cfg.Enabled {
continue
}
enabled[name] = struct{}{}
}
for name, cfg := range artifacts {
if !cfg.Enabled {
continue
}
for _, dep := range cfg.DependsOn {
trimmedDep := strings.TrimSpace(dep)
if _, ok := enabled[trimmedDep]; ok {
graph[name] = append(graph[name], trimmedDep)
}
}
}
visiting := make(map[string]bool, len(enabled))
visited := make(map[string]bool, len(enabled))
var visit func(node string) error
visit = func(node string) error {
if visiting[node] {
return fmt.Errorf("pipeline.scriptorium.artifacts enabled dependencies must not contain cycles")
}
if visited[node] {
return nil
}
visiting[node] = true
for _, dep := range graph[node] {
if err := visit(dep); err != nil {
return err
}
}
visiting[node] = false
visited[node] = true
return nil
}
for node := range enabled {
if err := visit(node); err != nil {
return err
}
}
return nil
}
func validateDuration(fieldName, value string) error {
trimmed := strings.TrimSpace(value)
if trimmed == "" {

View File

@@ -14,17 +14,26 @@ type ErrorRecord struct {
// InputRecord captures one resolved input and optional checksum.
type InputRecord struct {
Kind string `json:"kind"`
Path string `json:"path"`
Checksum string `json:"checksum,omitempty"`
Kind string `json:"kind"`
Path string `json:"path"`
Checksum string `json:"checksum,omitempty"`
Source string `json:"source,omitempty"`
S3Bucket string `json:"s3_bucket,omitempty"`
S3Key string `json:"s3_key,omitempty"`
S3Size int64 `json:"s3_size,omitempty"`
S3ETag string `json:"s3_etag,omitempty"`
SpoolPath string `json:"spool_path,omitempty"`
}
// ArtifactRecord captures one produced artifact and optional remote metadata.
type ArtifactRecord struct {
Kind string `json:"kind"`
SourceID string `json:"source_id,omitempty"`
LocalPath string `json:"local_path"`
RemoteKey string `json:"remote_key,omitempty"`
Checksum string `json:"checksum,omitempty"`
// ProducerRunID identifies the run that produced this durable artifact.
ProducerRunID string `json:"producer_run_id,omitempty"`
RemoteKey string `json:"remote_key,omitempty"`
Checksum string `json:"checksum,omitempty"`
}
// StageRecord tracks lifecycle and provenance for one pipeline stage.
@@ -45,6 +54,13 @@ type StageRecord struct {
// Manifest is the durable run-state record for a session execution.
type Manifest struct {
SessionID string `json:"session_id"`
Campaign string `json:"campaign,omitempty"`
RunID string `json:"run_id,omitempty"`
LocalWorkDir string `json:"local_workdir,omitempty"`
LocalSpoolDir string `json:"local_spool_dir,omitempty"`
S3Bucket string `json:"s3_bucket,omitempty"`
S3SessionPrefix string `json:"s3_session_prefix,omitempty"`
S3RunPrefix string `json:"s3_run_prefix,omitempty"`
PipelineVersion string `json:"pipeline_version,omitempty"`
CreatedAt time.Time `json:"created_at"`
UpdatedAt time.Time `json:"updated_at"`
@@ -107,6 +123,15 @@ func (m *Manifest) MarkStageSkipped(name string, at time.Time, reason string) {
m.UpdatedAt = at
}
// MarkStageStale marks a stage as stale so it is not skipped as idempotently complete.
func (m *Manifest) MarkStageStale(name string, at time.Time, reason string) {
s := m.ensureStage(name, at)
s.Status = StatusStale
s.Error = &ErrorRecord{Message: strings.TrimSpace(reason), Code: "stale", At: timePtr(at)}
s.UpdatedAt = at
m.UpdatedAt = at
}
func (m *Manifest) ensureStage(name string, at time.Time) *StageRecord {
if m.Stages == nil {
m.Stages = map[string]*StageRecord{}

View File

@@ -0,0 +1,164 @@
package manifest
import (
"strings"
"time"
)
type RunManifestStatus string
const (
RunManifestStatusRunning RunManifestStatus = "running"
RunManifestStatusSucceeded RunManifestStatus = "succeeded"
RunManifestStatusFailed RunManifestStatus = "failed"
)
type RunStageAction string
const (
RunStageActionRun RunStageAction = "run"
RunStageActionSkip RunStageAction = "skip"
)
// RunStageRecord tracks lifecycle and provenance for one stage within a single invocation.
type RunStageRecord struct {
Name string `json:"name"`
Action RunStageAction `json:"action"`
Status StageStatus `json:"status"`
CreatedAt time.Time `json:"created_at"`
UpdatedAt time.Time `json:"updated_at"`
StartedAt *time.Time `json:"started_at,omitempty"`
CompletedAt *time.Time `json:"completed_at,omitempty"`
Outputs []ArtifactRecord `json:"outputs,omitempty"`
Logs []string `json:"logs,omitempty"`
GeneratedConfigs []string `json:"generated_configs,omitempty"`
Error *ErrorRecord `json:"error,omitempty"`
Metadata map[string]any `json:"metadata,omitempty"`
}
// RunManifest is the invocation-scoped execution record under runs/{run_id}/manifest.json.
type RunManifest struct {
SessionID string `json:"session_id"`
Campaign string `json:"campaign,omitempty"`
RunID string `json:"run_id"`
Force bool `json:"force"`
RequestedStages []string `json:"requested_stages,omitempty"`
SessionManifestPath string `json:"session_manifest_path,omitempty"`
LocalWorkDir string `json:"local_workdir,omitempty"`
LocalSpoolDir string `json:"local_spool_dir,omitempty"`
S3Bucket string `json:"s3_bucket,omitempty"`
S3SessionPrefix string `json:"s3_session_prefix,omitempty"`
S3RunPrefix string `json:"s3_run_prefix,omitempty"`
CreatedAt time.Time `json:"created_at"`
UpdatedAt time.Time `json:"updated_at"`
StartedAt *time.Time `json:"started_at,omitempty"`
CompletedAt *time.Time `json:"completed_at,omitempty"`
Status RunManifestStatus `json:"status"`
LastError *ErrorRecord `json:"last_error,omitempty"`
Stages map[string]*RunStageRecord `json:"stages"`
Metadata map[string]any `json:"metadata,omitempty"`
}
// NewRun constructs a new run manifest with deterministic timestamps.
func NewRun(sessionID, campaign, runID string, force bool, requestedStages []string, now time.Time) *RunManifest {
return &RunManifest{
SessionID: strings.TrimSpace(sessionID),
Campaign: strings.TrimSpace(campaign),
RunID: strings.TrimSpace(runID),
Force: force,
RequestedStages: append([]string(nil), requestedStages...),
CreatedAt: now,
UpdatedAt: now,
StartedAt: timePtr(now),
Status: RunManifestStatusRunning,
Stages: map[string]*RunStageRecord{},
}
}
func (m *RunManifest) SetStageAction(name string, action RunStageAction, at time.Time) {
s := m.ensureStage(name, at)
s.Action = action
s.UpdatedAt = at
m.UpdatedAt = at
}
func (m *RunManifest) MarkStageRunning(name string, at time.Time) {
s := m.ensureStage(name, at)
s.Status = StatusRunning
s.StartedAt = timePtr(at)
s.CompletedAt = nil
s.Error = nil
s.UpdatedAt = at
m.UpdatedAt = at
}
func (m *RunManifest) MarkStageSucceeded(name string, at time.Time, outputs []ArtifactRecord) {
s := m.ensureStage(name, at)
s.Status = StatusSucceeded
s.CompletedAt = timePtr(at)
s.Error = nil
s.Outputs = append([]ArtifactRecord(nil), outputs...)
s.UpdatedAt = at
m.UpdatedAt = at
}
func (m *RunManifest) MarkStageFailed(name string, at time.Time, message string) {
s := m.ensureStage(name, at)
s.Status = StatusFailed
s.CompletedAt = timePtr(at)
s.Error = &ErrorRecord{Message: strings.TrimSpace(message), At: timePtr(at)}
s.UpdatedAt = at
m.LastError = &ErrorRecord{Message: strings.TrimSpace(message), At: timePtr(at)}
m.UpdatedAt = at
m.Status = RunManifestStatusFailed
m.CompletedAt = timePtr(at)
}
func (m *RunManifest) MarkStageSkipped(name string, at time.Time, reason string) {
s := m.ensureStage(name, at)
s.Status = StatusSkipped
s.CompletedAt = timePtr(at)
s.Error = &ErrorRecord{Message: strings.TrimSpace(reason), Code: "skipped", At: timePtr(at)}
s.UpdatedAt = at
m.UpdatedAt = at
}
func (m *RunManifest) MarkSucceeded(at time.Time) {
m.Status = RunManifestStatusSucceeded
m.CompletedAt = timePtr(at)
m.UpdatedAt = at
}
func (m *RunManifest) MarkFailed(at time.Time, message string) {
m.Status = RunManifestStatusFailed
m.CompletedAt = timePtr(at)
m.LastError = &ErrorRecord{Message: strings.TrimSpace(message), At: timePtr(at)}
m.UpdatedAt = at
}
func (m *RunManifest) ensureStage(name string, at time.Time) *RunStageRecord {
if m.Stages == nil {
m.Stages = map[string]*RunStageRecord{}
}
stageName := strings.TrimSpace(name)
s, ok := m.Stages[stageName]
if !ok || s == nil {
s = &RunStageRecord{
Name: stageName,
Action: RunStageActionRun,
Status: StatusPending,
CreatedAt: at,
UpdatedAt: at,
}
m.Stages[stageName] = s
}
if s.Name == "" {
s.Name = stageName
}
if s.CreatedAt.IsZero() {
s.CreatedAt = at
}
return s
}

Some files were not shown because too many files have changed in this diff Show More