Compare commits
18 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| c5c35cd3b4 | |||
| 574b1cde6c | |||
| ebb21b9201 | |||
| 958f446387 | |||
| 86caf4b222 | |||
| e38ed8ba97 | |||
| 3e79cf4724 | |||
| 859ae1ae10 | |||
| c63ecbab32 | |||
| 8480b74283 | |||
| 087869f7fa | |||
| 2b2a314d65 | |||
| 08b0f4edc5 | |||
| 571a289296 | |||
| 9f80635b42 | |||
| 11a3e174b6 | |||
| 9c5e5d6dc1 | |||
| c4e87f58c7 |
471
README.md
471
README.md
@@ -1,467 +1,22 @@
|
|||||||
# narratio
|
# narratio
|
||||||
|
|
||||||
`narratio` is a Go orchestration application for processing D&D session audio into transcripts and generated artifacts.
|
Narratio is a Go orchestration application that turns D&D session audio into polished transcripts and generated session artifacts.
|
||||||
|
|
||||||
## Current Implementation
|
It coordinates transcription, merge/polish/normalize/trim processing, artifact generation, archive publishing, and resumable run state in one operator workflow.
|
||||||
|
|
||||||
Implemented now:
|
|
||||||
|
|
||||||
- strict config loading/validation (`pipeline.yml` and `session.yml`)
|
|
||||||
- local workspace/session layout, locking, and manifest persistence
|
|
||||||
- resumable stage control (`run`, `plan`, `resume`, `run-stage`, `status`)
|
|
||||||
- real `prepare`, `transcribe`, `merge`, `polish`, `normalize`, `trim`, and `analyze` stages
|
|
||||||
- real WhisperX, Seriatim, and Audita adapters
|
|
||||||
- real Scriptorium subprocess adapter
|
|
||||||
- optional Scriptorium render diagnostics (`render_debug`)
|
|
||||||
|
|
||||||
Not implemented yet:
|
|
||||||
|
|
||||||
- `notify` stage behavior
|
|
||||||
- additional analyze artifacts beyond `session_recap`
|
|
||||||
- generic DAG orchestration
|
|
||||||
|
|
||||||
## Config Files
|
|
||||||
|
|
||||||
Narratio expects two YAML files:
|
|
||||||
|
|
||||||
- `pipeline.yml`: pipeline/workspace settings
|
|
||||||
- `session.yml`: per-session settings
|
|
||||||
|
|
||||||
Pipeline config lookup for CLI commands:
|
|
||||||
|
|
||||||
- if `--config <path>` is provided, Narratio uses that path
|
|
||||||
- if `--config` is omitted, Narratio searches in this order:
|
|
||||||
- `/usr/local/etc/narratio/pipeline.yml`
|
|
||||||
- `/etc/narratio/pipeline.yml`
|
|
||||||
|
|
||||||
Session config lookup for CLI commands:
|
|
||||||
|
|
||||||
- if `--session <path>` is provided, Narratio uses that path
|
|
||||||
- if `--session` is omitted, Narratio searches in this order:
|
|
||||||
- `./session.yml`
|
|
||||||
- `/usr/local/etc/narratio/session.yml`
|
|
||||||
- `/etc/narratio/session.yml`
|
|
||||||
|
|
||||||
Session template support:
|
|
||||||
|
|
||||||
- Narratio renders `session.yml` templates before strict YAML decode.
|
|
||||||
- `--session-id <value>` provides the `session_id` template variable.
|
|
||||||
- Supported placeholder forms:
|
|
||||||
- `{{session_id}}`
|
|
||||||
- `{{ session_id }}`
|
|
||||||
- unresolved template placeholders fail with a clear error.
|
|
||||||
- strict YAML validation still runs after rendering.
|
|
||||||
- concrete `session.yml` files without templates remain fully supported.
|
|
||||||
|
|
||||||
Optional secrets-from-files config:
|
|
||||||
|
|
||||||
- `pipeline.secrets.env_dir` may point to a directory of secret files
|
|
||||||
- each top-level file with an env-var-style name is loaded as an environment variable:
|
|
||||||
- file name = env var name
|
|
||||||
- file contents = env var value (trailing newline/CRLF trimmed)
|
|
||||||
- process environment wins: existing env vars are not overwritten
|
|
||||||
- if configured, Narratio fails fast when `env_dir` is missing/unreadable
|
|
||||||
- relative `env_dir` values resolve from Narratio’s current working directory
|
|
||||||
|
|
||||||
YAML decoding is strict (`KnownFields(true)`), so unknown fields fail fast.
|
|
||||||
|
|
||||||
Maintainer note: application defaults are centralized in [`internal/config/defaults.go`](internal/config/defaults.go).
|
|
||||||
|
|
||||||
## Storage And Archive Foundations
|
|
||||||
|
|
||||||
Narratio now includes configuration and path-model foundations for archive support, plus implemented prepare-stage S3 audio input.
|
|
||||||
|
|
||||||
Implemented foundations:
|
|
||||||
|
|
||||||
- `pipeline.storage.s3` config shape (`bucket`, `root_prefix`, `region`, `endpoint`, `force_path_style`, `access_key_id_env`, `secret_access_key_env`)
|
|
||||||
- `pipeline.spool` config shape (`root`, `delete_audio_after_archive`)
|
|
||||||
- `pipeline.archive` config shape (`enabled`, `upload_run`, `promote_artifacts`)
|
|
||||||
- promotion-rule validation (`from`/`to` required, relative-only paths, traversal rejected)
|
|
||||||
- `session.campaign` requirement for campaign-aware path construction
|
|
||||||
- optional `session.inputs.audio_s3.prefix` modeling and prepare-stage S3 audio download
|
|
||||||
- run ID generation and S3/local path helper foundations
|
|
||||||
- manifest run/path identity fields
|
|
||||||
|
|
||||||
Current defaults:
|
|
||||||
|
|
||||||
- `pipeline.storage.s3.root_prefix`: `dnd`
|
|
||||||
- `pipeline.storage.s3.access_key_id_env`: `OBJECT_STORAGE_KEY_ID`
|
|
||||||
- `pipeline.storage.s3.secret_access_key_env`: `OBJECT_STORAGE_KEY`
|
|
||||||
- `pipeline.workspace.cleanup_after_archive`: `false`
|
|
||||||
- `pipeline.spool.root`: `/var/spool/narratio`
|
|
||||||
- `pipeline.spool.delete_audio_after_archive`: `false`
|
|
||||||
- `pipeline.archive.enabled`: `true`
|
|
||||||
- `pipeline.archive.upload_run`: `true`
|
|
||||||
- default `pipeline.archive.promote_artifacts`:
|
|
||||||
- `transcripts/trimmed.json` -> `transcripts/trimmed.json` (`required: true`)
|
|
||||||
- `artifacts/session_recap.md` -> `artifacts/session_recap.md` (`required: true`)
|
|
||||||
|
|
||||||
Current boundaries:
|
|
||||||
|
|
||||||
- local development audio (`audio_dir` / `audio_files`) still works
|
|
||||||
- `audio_dir`/`audio_files` and `audio_s3` are mutually exclusive
|
|
||||||
- real S3-compatible backend now exists in the storage adapter package
|
|
||||||
- storage backend tests use fake storage and do not require live S3
|
|
||||||
- archive uploads successful run records under `runs/{run_id}/`
|
|
||||||
- archive does not upload local audio by default
|
|
||||||
- archive uploads promoted outputs to session-level keys using `archive.promote_artifacts`
|
|
||||||
- archive uploads `current/manifest.json`
|
|
||||||
- archive uploads `current/run_id.txt` last as the effective commit marker
|
|
||||||
- required missing promotions fail archive
|
|
||||||
- optional missing promotions are skipped and recorded
|
|
||||||
- cleanup remains conservative and opt-in:
|
|
||||||
- `pipeline.spool.delete_audio_after_archive: true` removes only the run-scoped spool audio directory after successful archive commit
|
|
||||||
- `pipeline.workspace.cleanup_after_archive: true` removes only the run-scoped local workdir after successful archive commit
|
|
||||||
- cleanup executes only after all selected stages for the command invocation succeed
|
|
||||||
- cleanup does not run for failed, incomplete, skipped, or unarchived runs
|
|
||||||
- local development `audio_dir`/`audio_files` source inputs are never deleted by spool cleanup
|
|
||||||
- S3 credentials are resolved from configured env-var names when both are present; if either is missing, Narratio falls back to the AWS SDK default credential chain
|
|
||||||
|
|
||||||
S3 input details and current boundaries are documented in [docs/s3-audio-input.md](docs/s3-audio-input.md).
|
|
||||||
|
|
||||||
## Remote Storage Backend
|
|
||||||
|
|
||||||
Narratio includes an object-store backend layer for future prepare/archive work:
|
|
||||||
|
|
||||||
- `List(ctx, prefix)`
|
|
||||||
- `Download(ctx, key, localPath)`
|
|
||||||
- `Upload(ctx, localPath, key, opts)`
|
|
||||||
- `Exists(ctx, key)`
|
|
||||||
|
|
||||||
Implemented backends:
|
|
||||||
|
|
||||||
- fake storage backend for deterministic tests
|
|
||||||
- S3-compatible backend built from `pipeline.storage.s3`
|
|
||||||
|
|
||||||
Key invariant:
|
|
||||||
|
|
||||||
- callers pass full bucket-relative object keys
|
|
||||||
- storage backends do not prepend `root_prefix` and do not infer session/campaign paths
|
|
||||||
|
|
||||||
Current boundary:
|
|
||||||
|
|
||||||
- `prepare` uses `List` + `Download` through the backend when `session.inputs.audio_s3` is configured
|
|
||||||
- `archive` uses `Upload` through the backend for successful run-record uploads
|
|
||||||
- `archive` also uses `Upload` for promotion writes and current pointers
|
|
||||||
- no failed or incomplete runs are uploaded
|
|
||||||
- local audio is not re-uploaded by default
|
|
||||||
|
|
||||||
Archive run-upload details and boundaries are documented in [docs/archive-storage.md](docs/archive-storage.md).
|
|
||||||
|
|
||||||
## Canonical Stage Order
|
|
||||||
|
|
||||||
1. `prepare`
|
|
||||||
2. `transcribe`
|
|
||||||
3. `merge`
|
|
||||||
4. `polish`
|
|
||||||
5. `normalize`
|
|
||||||
6. `trim`
|
|
||||||
7. `analyze`
|
|
||||||
8. `archive`
|
|
||||||
9. `notify`
|
|
||||||
|
|
||||||
## Transcript Tiers
|
|
||||||
|
|
||||||
- `transcripts/merged.json`: canonical deterministic merged transcript from Seriatim merge
|
|
||||||
- `transcripts/processed.json`: full raw Audita-polished transcript output
|
|
||||||
- `transcripts/normalized.json`: Seriatim-normalized transcript from the normalize stage
|
|
||||||
- `transcripts/trimmed.json`: gameplay-only normalized polished transcript from trim stage
|
|
||||||
|
|
||||||
## Seriatim Configuration
|
|
||||||
|
|
||||||
`pipeline.seriatim` configures the Seriatim subprocess adapter used by `merge`, `normalize`, and `trim`.
|
|
||||||
|
|
||||||
Minimal behavior:
|
|
||||||
|
|
||||||
- `pipeline.seriatim` may be omitted entirely.
|
|
||||||
- when omitted, Narratio defaults to:
|
|
||||||
- `binary: seriatim`
|
|
||||||
- `timeout: 10m`
|
|
||||||
- `output_schema: seriatim-intermediate`
|
|
||||||
- `coalesce_gap: 3.0`
|
|
||||||
- `report: true`
|
|
||||||
|
|
||||||
Optional overrides in `pipeline.seriatim` continue to work, including explicit binary paths and advanced `env` tuning values.
|
|
||||||
|
|
||||||
## Audita Configuration
|
|
||||||
|
|
||||||
`pipeline.audita` configures the real Audita subprocess adapter used by `polish`.
|
|
||||||
|
|
||||||
Minimal behavior:
|
|
||||||
|
|
||||||
- `pipeline.audita` may be omitted entirely.
|
|
||||||
- when omitted, Narratio defaults to:
|
|
||||||
- `binary: audita`
|
|
||||||
- `timeout: 3h`
|
|
||||||
- `report: true`
|
|
||||||
|
|
||||||
Optional:
|
|
||||||
|
|
||||||
- `llm_api_key_env` (when set, Narratio requires that env var and passes it to Audita as `AUDITA_LLM_API_KEY`)
|
|
||||||
- `modules` override list (when empty/omitted, Narratio does not pass `--modules`)
|
|
||||||
- `base_url` (when omitted, Narratio does not pass `--base-url`; Audita runtime defaults/config may apply)
|
|
||||||
- `model` (when omitted, Narratio does not pass `--model`; Audita runtime defaults/config may apply)
|
|
||||||
- `transcript_description`
|
|
||||||
- `config_path`
|
|
||||||
- `output_schema` (`bare-segments` or `audita-v1`)
|
|
||||||
- `work_dir_retention` (`always`, `auto`, or `never`)
|
|
||||||
- `total_llm_concurrency` (> 0 when provided)
|
|
||||||
- `proposal_llm_concurrency` (> 0 when provided)
|
|
||||||
- `validation_model`
|
|
||||||
- `validation_llm_concurrency` (> 0 when provided)
|
|
||||||
- `report` (defaults to `true`)
|
|
||||||
|
|
||||||
Narratio passes only configured optional Audita flags. Omitted optional values are left to Audita runtime defaults/config.
|
|
||||||
|
|
||||||
## Normalize Configuration
|
|
||||||
|
|
||||||
`pipeline.normalize` is optional. When omitted, Narratio defaults to:
|
|
||||||
|
|
||||||
- `output_path: transcripts/normalized.json`
|
|
||||||
- `output_schema: seriatim-intermediate`
|
|
||||||
- `report: true`
|
|
||||||
|
|
||||||
Allowed `normalize.output_schema` values:
|
|
||||||
|
|
||||||
- `seriatim-minimal`
|
|
||||||
- `seriatim-intermediate`
|
|
||||||
- `seriatim-full`
|
|
||||||
|
|
||||||
`normalize.output_path` is treated as session-workdir-relative when not absolute.
|
|
||||||
|
|
||||||
Normalize stage behavior summary:
|
|
||||||
|
|
||||||
- normalize runs after `polish` and before `trim`
|
|
||||||
- normalize resolves `transcripts/processed.json`
|
|
||||||
- normalize runs Seriatim `normalize` to produce `transcripts/normalized.json`
|
|
||||||
- normalize diagnostics are written to:
|
|
||||||
- `artifacts/seriatim.normalize.report.json` (when enabled)
|
|
||||||
- `logs/seriatim.normalize.stdout.log`
|
|
||||||
- `logs/seriatim.normalize.stderr.log`
|
|
||||||
- `config/seriatim.normalize.generated.yml`
|
|
||||||
|
|
||||||
## Trim Configuration
|
|
||||||
|
|
||||||
`pipeline.trim` is optional. If omitted, no trim config is loaded. If `trim.enabled` is omitted, it defaults to `false`.
|
|
||||||
|
|
||||||
When `trim.enabled: true`:
|
|
||||||
|
|
||||||
- `trim.output_path` is required
|
|
||||||
- `trim.bounds.prompt_id` is required
|
|
||||||
- `trim.bounds.transcript_input_name` is required
|
|
||||||
- `trim.bounds.output_path` is required
|
|
||||||
- `trim.bounds.timeout` must be a valid Go duration when provided
|
|
||||||
- `trim.bounds.render_debug: true` requires `trim.bounds.render_output_path`
|
|
||||||
- `trim.bounds.profile_id` may be empty to use the prompt default profile
|
|
||||||
|
|
||||||
Trim paths are treated as session-workdir-relative when not absolute.
|
|
||||||
|
|
||||||
Example trim config:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
trim:
|
|
||||||
enabled: true
|
|
||||||
output_path: "transcripts/trimmed.json"
|
|
||||||
bounds:
|
|
||||||
prompt_id: "dnd_session.bounds"
|
|
||||||
profile_id: ""
|
|
||||||
transcript_input_name: "transcript"
|
|
||||||
output_path: "artifacts/session_bounds.json"
|
|
||||||
timeout: "10m"
|
|
||||||
render_debug: false
|
|
||||||
render_output_path: "artifacts/session_bounds.render.json"
|
|
||||||
seriatim:
|
|
||||||
report: false
|
|
||||||
```
|
|
||||||
|
|
||||||
Trim behavior summary:
|
|
||||||
|
|
||||||
- trim discovers and validates `transcripts/normalized.json`
|
|
||||||
- trim uses Scriptorium bounds (`dnd_session.bounds` by example config) to produce `artifacts/session_bounds.json`
|
|
||||||
- bounds IDs are validated against the same normalized transcript ID space that Seriatim trim will consume
|
|
||||||
- trim converts bounds to Seriatim keep selector (for example `10-868`) and runs Seriatim trim
|
|
||||||
- if trim is disabled, Narratio copies normalized transcript to trimmed transcript and records `trim_action=copy_disabled`
|
|
||||||
|
|
||||||
Trim outputs and diagnostics:
|
|
||||||
|
|
||||||
- `artifacts/session_bounds.json`
|
|
||||||
- `transcripts/trimmed.json`
|
|
||||||
- `logs/scriptorium.bounds.stdout.log`
|
|
||||||
- `logs/scriptorium.bounds.stderr.log`
|
|
||||||
- `config/scriptorium.bounds.generated.yml`
|
|
||||||
- `logs/seriatim.trim.stdout.log`
|
|
||||||
- `logs/seriatim.trim.stderr.log`
|
|
||||||
- `config/seriatim.trim.generated.yml`
|
|
||||||
- optional bounds render-debug outputs:
|
|
||||||
- `artifacts/session_bounds.render.json`
|
|
||||||
- `logs/scriptorium.bounds.render.stdout.log`
|
|
||||||
- `logs/scriptorium.bounds.render.stderr.log`
|
|
||||||
- `config/scriptorium.bounds.render.generated.yml`
|
|
||||||
|
|
||||||
Render-debug files are diagnostics and are not treated as canonical stage output artifact refs.
|
|
||||||
|
|
||||||
## Scriptorium Configuration
|
|
||||||
|
|
||||||
`pipeline.scriptorium` is optional. When present, Narratio validates and uses it for analyze-stage artifact generation.
|
|
||||||
|
|
||||||
Key points:
|
|
||||||
|
|
||||||
- `scriptorium.binary` defaults to `scriptorium` when section is present
|
|
||||||
- `scriptorium.config_path` is optional
|
|
||||||
- `scriptorium.timeout` defaults to `10m` when omitted
|
|
||||||
- `scriptorium.render_debug` enables render diagnostics globally
|
|
||||||
- artifacts are configured under `scriptorium.artifacts` (map shape supports multiple artifacts)
|
|
||||||
- enabled artifacts require `prompt_id` and `output_path`
|
|
||||||
- artifact `render_debug` may override global render setting
|
|
||||||
- `vars` currently support boolean and string values
|
|
||||||
|
|
||||||
Example `session_recap` artifact definition:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
scriptorium:
|
|
||||||
binary: "scriptorium"
|
|
||||||
config_path: "/etc/scriptorium/config.yml"
|
|
||||||
timeout: "10m"
|
|
||||||
render_debug: false
|
|
||||||
|
|
||||||
artifacts:
|
|
||||||
session_recap:
|
|
||||||
enabled: true
|
|
||||||
prompt_id: "dnd.session_recap"
|
|
||||||
profile_id: "local-quality" # optional
|
|
||||||
output_path: "artifacts/session_recap.md"
|
|
||||||
timeout: "10m"
|
|
||||||
# render_debug: true # optional per-artifact override
|
|
||||||
|
|
||||||
inputs:
|
|
||||||
transcript:
|
|
||||||
source: "trimmed_transcript"
|
|
||||||
required: true
|
|
||||||
|
|
||||||
previous_recap:
|
|
||||||
source: "previous_session_artifact"
|
|
||||||
artifact: "session_recap"
|
|
||||||
path: "" # optional; set when available
|
|
||||||
required: false
|
|
||||||
|
|
||||||
vars:
|
|
||||||
session_id: true
|
|
||||||
session_date: true
|
|
||||||
campaign_name: true
|
|
||||||
previous_session_id: true
|
|
||||||
output_kind: "session_recap"
|
|
||||||
```
|
|
||||||
|
|
||||||
Prompt IDs and profile IDs are configuration values. They are not hardcoded in analyze-stage logic.
|
|
||||||
|
|
||||||
Do not put secrets in `pipeline.yml`. If API-key behavior is configured, use env var names only.
|
|
||||||
|
|
||||||
If `pipeline.secrets.env_dir` is configured, keep only references and secret files there; secret values are still not written to manifests, generated configs, or Narratio-managed logs.
|
|
||||||
|
|
||||||
## Scriptorium Runtime Behavior
|
|
||||||
|
|
||||||
Narratio integrates with Scriptorium through the public CLI subprocess contract:
|
|
||||||
|
|
||||||
- generation: `scriptorium run`
|
|
||||||
- diagnostics/testing: `scriptorium render --format json` when `render_debug` is enabled
|
|
||||||
|
|
||||||
For the initial implementation, only `session_recap` generation is supported.
|
|
||||||
|
|
||||||
Analyze-stage session recap behavior:
|
|
||||||
|
|
||||||
- preferred transcript artifact source IDs:
|
|
||||||
- `narratio.transcript.polished`
|
|
||||||
- `narratio.transcript.full`
|
|
||||||
- `narratio.transcript.trimmed`
|
|
||||||
- backward-compatible aliases remain supported:
|
|
||||||
- `processed_transcript`
|
|
||||||
- `normalized_transcript`
|
|
||||||
- `trimmed_transcript`
|
|
||||||
- session recap should use gameplay-only transcript input (`source: trimmed_transcript`)
|
|
||||||
- Narratio resolves transcript inputs from the artifact resolver (manifest producer outputs first, then canonical session paths)
|
|
||||||
- missing trimmed transcript fails clearly and advises running trim stage first
|
|
||||||
- `normalized_transcript` is the preferred full-transcript source for future table/meta-analysis artifacts
|
|
||||||
- `processed_transcript` remains supported for advanced/debug use cases
|
|
||||||
- optionally includes `previous_recap` when configured and resolvable
|
|
||||||
- omits optional previous recap when unavailable
|
|
||||||
- fails if required inputs are missing
|
|
||||||
- validates output file exists and is non-empty
|
|
||||||
|
|
||||||
Expected session output paths:
|
|
||||||
|
|
||||||
- `artifacts/session_recap.md`
|
|
||||||
- `logs/scriptorium.session_recap.stdout.log`
|
|
||||||
- `logs/scriptorium.session_recap.stderr.log`
|
|
||||||
- `config/scriptorium.session_recap.generated.yml`
|
|
||||||
- `artifacts/session_recap.render.json` when render diagnostics are enabled
|
|
||||||
|
|
||||||
## Examples
|
|
||||||
|
|
||||||
Starter files:
|
|
||||||
|
|
||||||
- `examples/pipeline.minimal.yml`
|
|
||||||
- `examples/pipeline.audita-overrides.yml`
|
|
||||||
- `examples/session.minimal.yml`
|
|
||||||
- `examples/session.template.yml`
|
|
||||||
- `examples/speakers.yml`
|
|
||||||
|
|
||||||
## Commands
|
|
||||||
|
|
||||||
Run tests:
|
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
go test ./...
|
narratio run --session-id 2026-04-04
|
||||||
```
|
```
|
||||||
|
|
||||||
Plan a run:
|
This command requires discoverable `pipeline.yml` and `session.yml` files (or explicit `--config` and `--session` flags).
|
||||||
|
|
||||||
```bash
|
## Documentation
|
||||||
go run ./cmd/narratio plan --session examples/session.minimal.yml
|
|
||||||
```
|
|
||||||
|
|
||||||
Use `--config <path>` to override default pipeline lookup when needed.
|
- [Configuration](docs/config.md)
|
||||||
|
- [CLI Reference](docs/cli.md)
|
||||||
Run with a discoverable session template:
|
- [Operations and Recovery](docs/operations.md)
|
||||||
|
- [Troubleshooting](docs/troubleshooting.md)
|
||||||
```bash
|
- [Development Guide](docs/development.md)
|
||||||
go run ./cmd/narratio run --session-id 2026-04-04
|
- [Architecture Principles](docs/architecture.md)
|
||||||
```
|
- [Internal Component Contracts](docs/internal/README.md)
|
||||||
|
- [Config Examples](examples/)
|
||||||
Run full pipeline:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
go run ./cmd/narratio run --config examples/pipeline.minimal.yml --session examples/session.minimal.yml
|
|
||||||
```
|
|
||||||
|
|
||||||
Run analyze only:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
go run ./cmd/narratio run-stage --config examples/pipeline.minimal.yml --session examples/session.minimal.yml analyze
|
|
||||||
```
|
|
||||||
|
|
||||||
Resume with a template session ID:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
go run ./cmd/narratio resume --config examples/pipeline.minimal.yml --session examples/session.template.yml --session-id 2026-04-04
|
|
||||||
```
|
|
||||||
|
|
||||||
## Operational Note
|
|
||||||
|
|
||||||
Checksum-based stale detection is not implemented yet.
|
|
||||||
|
|
||||||
If prepared inputs or prompt/runtime config change, rerun the appropriate upstream stages before relying on downstream artifacts.
|
|
||||||
|
|
||||||
Examples:
|
|
||||||
|
|
||||||
- glossary/autocorrect/speaker-context changes: rerun at least `merge`, `polish`, `normalize`, `trim`, and `analyze`
|
|
||||||
- trim bounds prompt/profile/config changes: rerun at least `normalize`, `trim`, and `analyze`
|
|
||||||
- session recap prompt/profile/input-source changes: rerun `analyze`
|
|
||||||
|
|
||||||
## Roadmap
|
|
||||||
|
|
||||||
Near-term roadmap:
|
|
||||||
|
|
||||||
- extend analyze to additional configured artifacts
|
|
||||||
- support workflows where later artifacts consume earlier generated artifacts
|
|
||||||
- keep orchestration explicit without a generic DAG engine
|
|
||||||
- implement archive and notify backends
|
|
||||||
|
|||||||
502
architecture.md
502
architecture.md
@@ -1,502 +0,0 @@
|
|||||||
# Narratio Architecture
|
|
||||||
|
|
||||||
## 1. Purpose
|
|
||||||
|
|
||||||
`narratio` is a Go orchestrator for D&D session processing. It runs a stage-based local pipeline from audio input through transcript processing and artifact generation, with manifest-based skip/force/resume behavior.
|
|
||||||
|
|
||||||
Narratio integrates with Scriptorium through the **public CLI** (`scriptorium run` and `scriptorium render`) via synchronous subprocess execution.
|
|
||||||
|
|
||||||
## 2. Current Status
|
|
||||||
|
|
||||||
Implemented:
|
|
||||||
|
|
||||||
- strict `pipeline.yml` + `session.yml` loading with strict YAML field checking (`KnownFields(true)`)
|
|
||||||
- local workspace/session layout, lock file handling, artifact path helpers, checksums, and atomic writes
|
|
||||||
- manifest store and stage status transitions for resumable runs
|
|
||||||
- real `prepare`, `transcribe`, `merge`, and `polish` stages
|
|
||||||
- real WhisperX HTTP adapter
|
|
||||||
- real Seriatim subprocess adapter
|
|
||||||
- real Audita subprocess adapter
|
|
||||||
- real Scriptorium subprocess adapter
|
|
||||||
- real `normalize` stage producing `transcripts/normalized.json`
|
|
||||||
- real `trim` stage producing `transcripts/trimmed.json`
|
|
||||||
- real `analyze` stage for initial `session_recap` generation
|
|
||||||
- optional Scriptorium render diagnostics (`render_debug`) before production run
|
|
||||||
- storage/archive configuration and validation foundations for:
|
|
||||||
- `pipeline.storage.s3`
|
|
||||||
- `pipeline.spool`
|
|
||||||
- `pipeline.archive` promotion rules
|
|
||||||
- `session.inputs.audio_s3`
|
|
||||||
- run identity and path-model foundations:
|
|
||||||
- run ID generation (`YYYYMMDDTHHMMSSZ-xxxxxxxx`)
|
|
||||||
- S3 session/run/current key builders
|
|
||||||
- campaign/session/run local work/spool path helpers
|
|
||||||
- manifest run/path identity fields (`campaign`, `run_id`, local and S3 prefixes)
|
|
||||||
- remote storage backend layer:
|
|
||||||
- narrow object-store interface (`List`, `Download`, `Upload`, `Exists`)
|
|
||||||
- fake storage backend for deterministic tests (no network dependency)
|
|
||||||
- S3-compatible backend using AWS SDK v2
|
|
||||||
- config-based object-store construction helper
|
|
||||||
|
|
||||||
Still placeholder/future:
|
|
||||||
|
|
||||||
- `notify` stage behavior
|
|
||||||
- additional Scriptorium artifact types beyond `session_recap`
|
|
||||||
- artifact-to-artifact workflows beyond the initial single-artifact implementation
|
|
||||||
- generic stale detection based on input/config checksums
|
|
||||||
|
|
||||||
## 3. Pipeline and Stage Boundaries
|
|
||||||
|
|
||||||
Canonical stage order:
|
|
||||||
|
|
||||||
1. `prepare`
|
|
||||||
2. `transcribe`
|
|
||||||
3. `merge`
|
|
||||||
4. `polish`
|
|
||||||
5. `normalize`
|
|
||||||
6. `trim`
|
|
||||||
7. `analyze`
|
|
||||||
8. `archive`
|
|
||||||
9. `notify`
|
|
||||||
|
|
||||||
Boundary rules:
|
|
||||||
|
|
||||||
- orchestration logic lives in `internal/app`
|
|
||||||
- stage business logic lives in `internal/stage`
|
|
||||||
- external-tool CLI construction lives in adapter packages
|
|
||||||
- Scriptorium CLI details stay in `internal/adapters/scriptorium`
|
|
||||||
|
|
||||||
## 4. Scriptorium Integration Model
|
|
||||||
|
|
||||||
Integration mode:
|
|
||||||
|
|
||||||
- public CLI subprocesses only (no Scriptorium internal Go packages, no HTTP API)
|
|
||||||
- production generation uses `scriptorium run`
|
|
||||||
- diagnostics/testing render uses `scriptorium render --format json`
|
|
||||||
|
|
||||||
Run invocation shape used by adapter:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
scriptorium run --prompt <prompt_id> --input name=path --out <output_path>
|
|
||||||
```
|
|
||||||
|
|
||||||
Optional flags passed when configured:
|
|
||||||
|
|
||||||
- `--config <path>`
|
|
||||||
- `--profile <profile_id>`
|
|
||||||
- repeated `--var name=value`
|
|
||||||
- repeated `--input name=path`
|
|
||||||
- `--timeout <duration>`
|
|
||||||
- `--api-key-env <ENV_NAME>` when configured
|
|
||||||
|
|
||||||
Render invocation shape used by adapter:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
scriptorium render --prompt <prompt_id> --input name=path --format json --out <render_output_path>
|
|
||||||
```
|
|
||||||
|
|
||||||
Adapter behavior:
|
|
||||||
|
|
||||||
- always passes `--out`
|
|
||||||
- captures stdout/stderr separately
|
|
||||||
- writes generated invocation metadata YAML (redacted, no secrets)
|
|
||||||
- treats exit code `0` as success
|
|
||||||
- treats exit code `1` as failure
|
|
||||||
- treats exit code `2` as failure with `validation_failed=true` and preserves output metadata when available
|
|
||||||
- validates successful output files exist and are non-empty
|
|
||||||
- does not treat non-empty stderr as failure by itself
|
|
||||||
|
|
||||||
## 5. Configuration Contract
|
|
||||||
|
|
||||||
CLI pipeline config path resolution:
|
|
||||||
|
|
||||||
- when `--config <path>` is provided, that path is used
|
|
||||||
- when `--config` is omitted, Narratio searches defaults in order:
|
|
||||||
- `/usr/local/etc/narratio/pipeline.yml`
|
|
||||||
- `/etc/narratio/pipeline.yml`
|
|
||||||
- default values are centralized in `internal/config/defaults.go`
|
|
||||||
|
|
||||||
CLI session config path resolution:
|
|
||||||
|
|
||||||
- when `--session <path>` is provided, that path is used
|
|
||||||
- when `--session` is omitted, Narratio searches defaults in order:
|
|
||||||
- `./session.yml`
|
|
||||||
- `/usr/local/etc/narratio/session.yml`
|
|
||||||
- `/etc/narratio/session.yml`
|
|
||||||
|
|
||||||
Session template rendering:
|
|
||||||
|
|
||||||
- session templates are rendered before strict YAML decode
|
|
||||||
- `--session-id <value>` provides the `session_id` template variable
|
|
||||||
- supported placeholders:
|
|
||||||
- `{{session_id}}`
|
|
||||||
- `{{ session_id }}`
|
|
||||||
- unresolved placeholders fail clearly
|
|
||||||
- strict `KnownFields(true)` YAML validation still applies after rendering
|
|
||||||
- if rendered `session.session_id` conflicts with `--session-id`, load fails clearly
|
|
||||||
|
|
||||||
Optional pipeline secrets directory:
|
|
||||||
|
|
||||||
- `pipeline.secrets.env_dir` enables loading environment variables from local files before command execution
|
|
||||||
- file name = env var name; file contents = env var value (trailing newline/CRLF trimmed)
|
|
||||||
- only env-var-style file names are considered; other entries are ignored
|
|
||||||
- existing process environment values are preserved (not overwritten)
|
|
||||||
- if configured, unreadable/missing `env_dir` fails command execution early
|
|
||||||
- relative `env_dir` values are resolved from current working directory
|
|
||||||
|
|
||||||
Storage and archive foundations:
|
|
||||||
|
|
||||||
- `pipeline.storage.s3` is available for modeling S3 coordinates:
|
|
||||||
- `bucket`
|
|
||||||
- `root_prefix` (default `dnd`)
|
|
||||||
- `region`
|
|
||||||
- `endpoint`
|
|
||||||
- `force_path_style` (default `false`)
|
|
||||||
- `access_key_id_env` (default `OBJECT_STORAGE_KEY_ID`)
|
|
||||||
- `secret_access_key_env` (default `OBJECT_STORAGE_KEY`)
|
|
||||||
- `pipeline.spool.root` defaults to `/var/spool/narratio`
|
|
||||||
- `pipeline.workspace.cleanup_after_archive` defaults to `false`
|
|
||||||
- `pipeline.spool.delete_audio_after_archive` defaults to `false`
|
|
||||||
- `pipeline.archive` is optional and defaults to:
|
|
||||||
- `enabled: true`
|
|
||||||
- `upload_run: true`
|
|
||||||
- default `promote_artifacts`:
|
|
||||||
- `transcripts/trimmed.json`
|
|
||||||
- `artifacts/session_recap.md`
|
|
||||||
- archive promotion rules enforce safe relative paths:
|
|
||||||
- `from` and `to` are required
|
|
||||||
- absolute paths are rejected
|
|
||||||
- traversal segments such as `..` are rejected
|
|
||||||
|
|
||||||
Session input foundations:
|
|
||||||
|
|
||||||
- `session.campaign` is required
|
|
||||||
- local audio remains supported through `session.inputs.audio_dir` or `session.inputs.audio_files`
|
|
||||||
- optional S3 audio input shape is `session.inputs.audio_s3.prefix`
|
|
||||||
- `audio_dir`/`audio_files` and `audio_s3` are mutually exclusive
|
|
||||||
- when `audio_s3` is configured, `prepare` lists and downloads `.flac` objects through the object-store backend
|
|
||||||
|
|
||||||
Cross-config validation scope:
|
|
||||||
|
|
||||||
- `pipeline.storage.s3.bucket` is required only when an S3-dependent feature is explicitly configured (for current foundations, that includes `session.inputs.audio_s3`, and archive upload intent when using `storage.backend: s3`)
|
|
||||||
- no AWS credential values are stored in Narratio config; only env-var names are configured
|
|
||||||
- when both configured credential env vars resolve to non-empty values, the S3 backend uses them as static credentials
|
|
||||||
- when either configured credential value is missing, the S3 backend falls back to the AWS SDK default credential chain
|
|
||||||
|
|
||||||
Remote object-store backend scope:
|
|
||||||
|
|
||||||
- remote storage APIs are isolated to `internal/adapters/storage`
|
|
||||||
- AWS SDK types remain contained within the S3 backend implementation package
|
|
||||||
- S3 key/session path semantics remain outside the backend, with this invariant:
|
|
||||||
- callers pass full bucket-relative object keys
|
|
||||||
- backend methods do not prepend `root_prefix` or infer campaign/session/run paths
|
|
||||||
- `prepare` now uses object-store `List` and `Download` for S3 audio input
|
|
||||||
- `archive` now uses object-store `Upload` for successful run-record upload under the run prefix
|
|
||||||
- `archive` now uses object-store `Upload` for promoted outputs and current pointers
|
|
||||||
|
|
||||||
Prepare S3 audio behavior (implemented):
|
|
||||||
|
|
||||||
- compute session prefix as `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/`
|
|
||||||
- resolve `session.inputs.audio_s3.prefix` under that session prefix
|
|
||||||
- list objects under the computed audio prefix and filter `.flac` keys
|
|
||||||
- fail clearly when no `.flac` objects are found
|
|
||||||
- download selected objects to spool audio path:
|
|
||||||
- `{spool.root}/{campaign}/{session_id}/{run_id}/audio/`
|
|
||||||
- materialize audio files into workdir audio path:
|
|
||||||
- `{workspace.root}/work/{campaign}/{session_id}/{run_id}/audio/`
|
|
||||||
- record S3 provenance in manifest input records (bucket/key/metadata/local paths/checksum)
|
|
||||||
- no AWS SDK types are used in stage code; storage implementation details stay in storage adapter packages
|
|
||||||
|
|
||||||
Archive publishing behavior (implemented):
|
|
||||||
|
|
||||||
- `archive` verifies prerequisite stage success before upload:
|
|
||||||
- `prepare`, `transcribe`, `merge`, `polish`, `normalize`, `trim`, `analyze`
|
|
||||||
- only successful/completed runs are uploaded
|
|
||||||
- uploaded run record destination is:
|
|
||||||
- `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/runs/{run_id}/`
|
|
||||||
- uploaded existing local paths include:
|
|
||||||
- `inputs/`, `transcripts/`, `artifacts/`, optional `reports/`, `config/`, `logs/`, and `manifest.json`
|
|
||||||
- local `audio/` is intentionally excluded from upload by default
|
|
||||||
- file upload order is deterministic (sorted relative paths)
|
|
||||||
- `archive.enabled: false` and `archive.upload_run: false` skip upload cleanly
|
|
||||||
- stage metadata records non-secret upload context:
|
|
||||||
- run upload details, promoted output details, current manifest key, current pointer key
|
|
||||||
- no secrets, transcript contents, prompt contents, or environment dumps
|
|
||||||
- promotion rules:
|
|
||||||
- `from` resolves from local workdir
|
|
||||||
- `to` resolves under session-level S3 root
|
|
||||||
- missing required source fails archive
|
|
||||||
- missing optional source is skipped and recorded
|
|
||||||
- default promoted outputs:
|
|
||||||
- `transcripts/trimmed.json`
|
|
||||||
- `artifacts/session_recap.md`
|
|
||||||
- current pointers:
|
|
||||||
- `current/manifest.json` uploaded after run upload and promotions
|
|
||||||
- `current/run_id.txt` uploaded last with `{run_id}\n`
|
|
||||||
- `current/run_id.txt` is the effective commit marker
|
|
||||||
- if promotion or current-manifest upload fails, archive returns failure and does not write `current/run_id.txt`
|
|
||||||
- failed/incomplete runs remain local and are not uploaded
|
|
||||||
- post-archive local cleanup (implemented, opt-in):
|
|
||||||
- cleanup runs only after archive succeeded and wrote `current/run_id.txt`
|
|
||||||
- cleanup is executed after all selected stages in the command invocation succeed (for example, a later `notify` failure leaves local files intact)
|
|
||||||
- `pipeline.spool.delete_audio_after_archive: true` removes only `{spool.root}/{campaign}/{session_id}/{run_id}/audio/`
|
|
||||||
- `pipeline.workspace.cleanup_after_archive: true` removes only `{workspace.root}/work/{campaign}/{session_id}/{run_id}/`
|
|
||||||
- cleanup does not run when archive is skipped/disabled/fails or when run upload is disabled
|
|
||||||
- local development `audio_dir`/`audio_files` inputs are never removed by spool cleanup
|
|
||||||
|
|
||||||
`pipeline.scriptorium` is optional. Existing pipelines without Scriptorium continue to work.
|
|
||||||
|
|
||||||
`pipeline.trim` is optional. Existing pipelines without trim config continue to work.
|
|
||||||
|
|
||||||
`pipeline.normalize` is optional. Existing pipelines without normalize config continue to work.
|
|
||||||
|
|
||||||
`pipeline.audita` drives the real Audita subprocess adapter for the `polish` stage.
|
|
||||||
|
|
||||||
Audita defaulted fields:
|
|
||||||
|
|
||||||
- `binary` defaults to `audita`
|
|
||||||
- `timeout` defaults to `3h`
|
|
||||||
- `report` defaults to `true`
|
|
||||||
|
|
||||||
Audita optional fields:
|
|
||||||
|
|
||||||
- `llm_api_key_env` (enforced only when configured)
|
|
||||||
- `modules` override list (when omitted/empty, Narratio does not pass `--modules`)
|
|
||||||
- `base_url` (when omitted, Narratio does not pass `--base-url`)
|
|
||||||
- `model` (when omitted, Narratio does not pass `--model`)
|
|
||||||
- `transcript_description`
|
|
||||||
- `config_path`
|
|
||||||
- `output_schema` (`bare-segments` or `audita-v1`)
|
|
||||||
- `work_dir_retention` (`always`, `auto`, `never`)
|
|
||||||
- `total_llm_concurrency` (> 0 when provided)
|
|
||||||
- `proposal_llm_concurrency` (> 0 when provided)
|
|
||||||
- `validation_model`
|
|
||||||
- `validation_llm_concurrency` (> 0 when provided)
|
|
||||||
- `report` override
|
|
||||||
|
|
||||||
Narratio passes only configured optional Audita flags; omitted optional values defer to Audita runtime defaults/config.
|
|
||||||
|
|
||||||
Seriatim defaults:
|
|
||||||
|
|
||||||
- `pipeline.seriatim` may be omitted
|
|
||||||
- `binary` defaults to `seriatim`
|
|
||||||
- `timeout` defaults to `10m`
|
|
||||||
- `output_schema` defaults to `seriatim-intermediate`
|
|
||||||
- `coalesce_gap` defaults to `3.0`
|
|
||||||
- `report` defaults to `true`
|
|
||||||
|
|
||||||
When `pipeline.normalize` is omitted, defaults are applied:
|
|
||||||
|
|
||||||
- `output_path: transcripts/normalized.json`
|
|
||||||
- `output_schema: seriatim-intermediate`
|
|
||||||
- `report: true`
|
|
||||||
|
|
||||||
When `pipeline.normalize` is present:
|
|
||||||
|
|
||||||
- `output_path` must be non-empty
|
|
||||||
- `output_schema` must be one of `seriatim-minimal`, `seriatim-intermediate`, or `seriatim-full`
|
|
||||||
- relative `output_path` values are session-workdir-relative paths
|
|
||||||
- Seriatim binary settings still come from `pipeline.seriatim`
|
|
||||||
|
|
||||||
When `pipeline.trim` is present:
|
|
||||||
|
|
||||||
- `enabled` is optional and defaults to `false` when omitted
|
|
||||||
- relative `output_path`, `bounds.output_path`, and `bounds.render_output_path` values are session-workdir-relative paths
|
|
||||||
- do not store secrets in trim config values
|
|
||||||
|
|
||||||
When `pipeline.trim.enabled: true`:
|
|
||||||
|
|
||||||
- `output_path` is required and non-empty
|
|
||||||
- `bounds.prompt_id` is required and non-empty
|
|
||||||
- `bounds.transcript_input_name` is required and non-empty
|
|
||||||
- `bounds.output_path` is required and non-empty
|
|
||||||
- `bounds.timeout` must parse as a Go duration when provided
|
|
||||||
- `bounds.render_debug: true` requires non-empty `bounds.render_output_path`
|
|
||||||
- `bounds.profile_id` may be empty to use the prompt default profile
|
|
||||||
- prompt IDs are config values, not hardcoded stage logic
|
|
||||||
|
|
||||||
When `pipeline.scriptorium` is present:
|
|
||||||
|
|
||||||
- `binary` defaults to `scriptorium` when omitted
|
|
||||||
- `config_path` is optional; when provided it must be non-empty
|
|
||||||
- `timeout` is optional; when provided it must parse as a Go duration
|
|
||||||
- default `timeout` is `10m`
|
|
||||||
- unknown YAML fields fail strict decode
|
|
||||||
|
|
||||||
Artifacts are configured as a map under `pipeline.scriptorium.artifacts` so multiple artifacts are possible in the config shape.
|
|
||||||
|
|
||||||
For each artifact definition:
|
|
||||||
|
|
||||||
- `enabled: true` requires non-empty `prompt_id`
|
|
||||||
- `enabled: true` requires non-empty `output_path`
|
|
||||||
- `timeout` must parse as Go duration when present
|
|
||||||
- optional per-artifact `render_debug` may override global `scriptorium.render_debug`
|
|
||||||
- `inputs` are named and each input requires non-empty `source`
|
|
||||||
- inputs may be optional (`required: false`)
|
|
||||||
- `vars` values currently support `string` and `bool`
|
|
||||||
|
|
||||||
Prompt IDs and profile IDs are configuration values, not hardcoded stage logic.
|
|
||||||
|
|
||||||
Trim config shape:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
trim:
|
|
||||||
enabled: true
|
|
||||||
output_path: "transcripts/trimmed.json"
|
|
||||||
bounds:
|
|
||||||
prompt_id: "dnd_session.bounds"
|
|
||||||
profile_id: ""
|
|
||||||
transcript_input_name: "transcript"
|
|
||||||
output_path: "artifacts/session_bounds.json"
|
|
||||||
timeout: "10m"
|
|
||||||
render_debug: false
|
|
||||||
render_output_path: "artifacts/session_bounds.render.json"
|
|
||||||
seriatim:
|
|
||||||
report: false
|
|
||||||
```
|
|
||||||
|
|
||||||
## 6. Transcript Tiers
|
|
||||||
|
|
||||||
Narratio currently produces and uses four transcript tiers:
|
|
||||||
|
|
||||||
- `transcripts/merged.json`: canonical deterministic merged transcript from Seriatim merge
|
|
||||||
- `transcripts/processed.json`: full raw Audita-polished transcript output (includes pre/post-game content)
|
|
||||||
- `transcripts/normalized.json`: normalized transcript generated by Seriatim normalize
|
|
||||||
- `transcripts/trimmed.json`: gameplay-only normalized polished transcript from trim stage
|
|
||||||
|
|
||||||
Trim reads `transcripts/normalized.json`, validates bounds IDs against that same transcript ID space, and writes `transcripts/trimmed.json`.
|
|
||||||
|
|
||||||
## 7. Normalize Stage (Current Implementation)
|
|
||||||
|
|
||||||
Normalize stage behavior:
|
|
||||||
|
|
||||||
- stage order position: after `polish` and before `trim`
|
|
||||||
- discovers processed transcript from manifest polish outputs (`transcript_processed`) when present, else `work/<session_id>/transcripts/processed.json`
|
|
||||||
- validates processed transcript JSON shape (`segments` array required)
|
|
||||||
- runs Seriatim `normalize` to produce normalized transcript
|
|
||||||
- validates normalized transcript JSON shape (`segments` array required)
|
|
||||||
- validates normalize report JSON when enabled
|
|
||||||
|
|
||||||
Expected normalize outputs and diagnostics:
|
|
||||||
|
|
||||||
- `transcripts/normalized.json`
|
|
||||||
- `artifacts/seriatim.normalize.report.json` (when normalize report is enabled)
|
|
||||||
- `logs/seriatim.normalize.stdout.log`
|
|
||||||
- `logs/seriatim.normalize.stderr.log`
|
|
||||||
- `config/seriatim.normalize.generated.yml`
|
|
||||||
|
|
||||||
## 8. Trim Stage (Current Implementation)
|
|
||||||
|
|
||||||
Trim stage behavior:
|
|
||||||
|
|
||||||
- stage order position: after `normalize` and before `analyze`
|
|
||||||
- discovers normalized transcript from manifest normalize outputs (`transcript_normalized`) when present, else `work/<session_id>/transcripts/normalized.json`
|
|
||||||
- validates normalized transcript JSON shape (`segments` array required)
|
|
||||||
- when `trim.enabled: false` (or trim config omitted), deterministically copies normalized transcript to `transcripts/trimmed.json` and records `trim_action=copy_disabled`
|
|
||||||
- when `trim.enabled: true`:
|
|
||||||
- runs Scriptorium bounds prompt using configured `trim.bounds.prompt_id`
|
|
||||||
- writes bounds output to configured path (typically `artifacts/session_bounds.json`)
|
|
||||||
- parses and validates bounds output against the same normalized transcript being trimmed
|
|
||||||
- converts bounds range to Seriatim keep selector (for example `10-868`)
|
|
||||||
- runs Seriatim `trim` to produce `transcripts/trimmed.json`
|
|
||||||
- supports no-trim bounds actions (`none`/`copy`) by copying normalized transcript unchanged
|
|
||||||
- validates trimmed transcript JSON shape (`segments` array required)
|
|
||||||
|
|
||||||
Expected trim outputs and diagnostics:
|
|
||||||
|
|
||||||
- `artifacts/session_bounds.json`
|
|
||||||
- `transcripts/trimmed.json`
|
|
||||||
- `logs/scriptorium.bounds.stdout.log`
|
|
||||||
- `logs/scriptorium.bounds.stderr.log`
|
|
||||||
- `config/scriptorium.bounds.generated.yml`
|
|
||||||
- `logs/seriatim.trim.stdout.log`
|
|
||||||
- `logs/seriatim.trim.stderr.log`
|
|
||||||
- `config/seriatim.trim.generated.yml`
|
|
||||||
- optional bounds render-debug outputs when enabled:
|
|
||||||
- `artifacts/session_bounds.render.json`
|
|
||||||
- `logs/scriptorium.bounds.render.stdout.log`
|
|
||||||
- `logs/scriptorium.bounds.render.stderr.log`
|
|
||||||
- `config/scriptorium.bounds.render.generated.yml`
|
|
||||||
|
|
||||||
Render-debug files are diagnostics. They are recorded in stage metadata/log/config refs and are not treated as canonical stage output artifact refs.
|
|
||||||
|
|
||||||
## 9. Analyze Stage (Current Implementation)
|
|
||||||
|
|
||||||
The current real analyze implementation supports only `scriptorium.artifacts.session_recap`.
|
|
||||||
|
|
||||||
Behavior:
|
|
||||||
|
|
||||||
- if `pipeline.scriptorium` is missing, analyze returns a skipped result with metadata
|
|
||||||
- if no Scriptorium artifacts are enabled, analyze returns a skipped result with metadata
|
|
||||||
- if enabled artifacts exist but `session_recap` is not enabled, analyze fails clearly
|
|
||||||
- available transcript input sources for configured artifacts: `processed_transcript`, `normalized_transcript`, `trimmed_transcript`
|
|
||||||
- `session_recap` should use `trimmed_transcript` input (`transcripts/trimmed.json`) for in-universe recap generation
|
|
||||||
- `trimmed_transcript` input is resolved from manifest (`trim` output kind `transcript_trimmed`) when available, otherwise fallback path `work/<session_id>/transcripts/trimmed.json`
|
|
||||||
- `normalized_transcript` input is resolved from manifest (`normalize` output kind `transcript_normalized`) when available, otherwise fallback path `work/<session_id>/transcripts/normalized.json`
|
|
||||||
- `processed_transcript` input is resolved from manifest (`polish` output kind `transcript_processed`) when available, otherwise fallback path `work/<session_id>/transcripts/processed.json`
|
|
||||||
- `normalized_transcript` is the preferred full-transcript source for future table/meta-analysis artifacts
|
|
||||||
- `processed_transcript` remains available for advanced/debug use cases
|
|
||||||
- transcript inputs are validated as JSON with top-level `segments` array
|
|
||||||
- configured inputs are resolved by source
|
|
||||||
- optional `previous_recap` is omitted when unavailable
|
|
||||||
- required `previous_recap` fails before invocation when unavailable
|
|
||||||
- vars are built from config + session metadata
|
|
||||||
- `render_debug` controls pre-run `scriptorium render` diagnostics
|
|
||||||
- render failure stops stage before production run
|
|
||||||
- render output is validated as JSON
|
|
||||||
- production call uses Scriptorium adapter `RunArtifact`
|
|
||||||
- successful run output must exist and be non-empty
|
|
||||||
- missing `trimmed_transcript` input for configured `trimmed_transcript` source fails clearly with guidance to run trim stage first
|
|
||||||
- manifest records output refs, logs, generated config paths, and non-secret provenance metadata
|
|
||||||
|
|
||||||
## 10. Session Recap Paths
|
|
||||||
|
|
||||||
Current expected paths for `session_recap`:
|
|
||||||
|
|
||||||
- artifact output: `artifacts/session_recap.md`
|
|
||||||
- run stdout log: `logs/scriptorium.session_recap.stdout.log`
|
|
||||||
- run stderr log: `logs/scriptorium.session_recap.stderr.log`
|
|
||||||
- run generated invocation/config: `config/scriptorium.session_recap.generated.yml`
|
|
||||||
- render output (when enabled): `artifacts/session_recap.render.json`
|
|
||||||
- render stdout log: `logs/scriptorium.session_recap.render.stdout.log`
|
|
||||||
- render stderr log: `logs/scriptorium.session_recap.render.stderr.log`
|
|
||||||
- render generated invocation/config: `config/scriptorium.session_recap.render.generated.yml`
|
|
||||||
|
|
||||||
## 11. Security and Privacy
|
|
||||||
|
|
||||||
- do not store secrets in pipeline YAML, generated invocation YAML, logs, or manifest metadata
|
|
||||||
- if API-key integration is configured, pass env var names only (never raw key values)
|
|
||||||
- with `pipeline.secrets.env_dir`, secret file values are loaded into process env only and are not persisted in manifest metadata or generated configs
|
|
||||||
- avoid logging transcript content or rendered prompt content by default
|
|
||||||
- treat generated artifacts and logs as potentially sensitive session material
|
|
||||||
|
|
||||||
## 12. Operational Caveat (Pre-Stale-Detection)
|
|
||||||
|
|
||||||
Checksum-based stale detection is not implemented yet.
|
|
||||||
|
|
||||||
If prepared inputs or prompt/runtime configuration change (for example glossary files, prompt IDs, profile IDs, or relevant pipeline settings), rerun the appropriate prior stages to refresh downstream artifacts.
|
|
||||||
|
|
||||||
Examples:
|
|
||||||
|
|
||||||
- glossary or autocorrect changes usually require rerunning at least `merge`, `polish`, `normalize`, `trim`, and `analyze`
|
|
||||||
- trim prompt/profile changes require rerunning at least `normalize`, `trim`, and `analyze`
|
|
||||||
- session recap prompt/profile/input-source changes require rerunning `analyze`
|
|
||||||
|
|
||||||
## 13. Roadmap
|
|
||||||
|
|
||||||
Planned next steps:
|
|
||||||
|
|
||||||
- extend analyze beyond `session_recap` to additional configured artifacts
|
|
||||||
- support artifact inputs that consume prior generated artifacts
|
|
||||||
- keep this composable without adding a generic DAG engine in the near term
|
|
||||||
- implement real `archive` backend behavior
|
|
||||||
- implement real `notify` backend behavior
|
|
||||||
- add checksum-based stale detection and stale transitions
|
|
||||||
|
|
||||||
Architectural invariants remain:
|
|
||||||
|
|
||||||
- strict config decoding/validation
|
|
||||||
- manifest-driven run control
|
|
||||||
- clear stage/adapter separation
|
|
||||||
- configuration-driven prompt/profile/input/vars/output mapping
|
|
||||||
- Scriptorium integration through public CLI subprocess contract
|
|
||||||
202
docs/architecture.md
Normal file
202
docs/architecture.md
Normal file
@@ -0,0 +1,202 @@
|
|||||||
|
# Narratio Architecture
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
|
||||||
|
`narratio` is a Go orchestration application for processing D&D session audio into polished transcripts and generated session artifacts.
|
||||||
|
|
||||||
|
This document defines the development principles for the project. It is inward-facing: its audience is developers and LLM coding agents. It should guide future changes, not serve as a complete implementation reference.
|
||||||
|
|
||||||
|
Implemented component details belong under `docs/internal/`.
|
||||||
|
|
||||||
|
## Project Shape
|
||||||
|
|
||||||
|
Narratio is a modular, stage-driven orchestrator.
|
||||||
|
|
||||||
|
It coordinates specialized downstream systems rather than reimplementing their domains:
|
||||||
|
|
||||||
|
- WhisperX handles transcription.
|
||||||
|
- Seriatim handles deterministic transcript merge/normalization/trim behavior.
|
||||||
|
- Audita handles transcript correction and polishing.
|
||||||
|
- Scriptorium handles prompt execution and generated artifacts.
|
||||||
|
|
||||||
|
Narratio owns orchestration, configuration loading, session/run state, local and remote path modeling, manifest persistence, stage sequencing, resume behavior, and archive semantics.
|
||||||
|
|
||||||
|
Narratio should remain explicit and comprehensible. It is not intended to become a generic workflow engine.
|
||||||
|
|
||||||
|
## Core Principles
|
||||||
|
|
||||||
|
### Modular and composable
|
||||||
|
|
||||||
|
Code should be organized around clear responsibilities. Stages, adapters, config loading, manifest persistence, path construction, and storage behavior should remain separable and independently testable.
|
||||||
|
|
||||||
|
### Hexagonal boundaries
|
||||||
|
|
||||||
|
External systems should be isolated behind narrow adapters. Stage logic should depend on Narratio-level interfaces and data structures, not on external SDK types, subprocess argument construction, or transport-specific details.
|
||||||
|
|
||||||
|
### Standard library preference
|
||||||
|
|
||||||
|
Prefer the Go standard library. Add dependencies only when they provide substantial value, are necessary for an external integration, or are a widely used de facto standard.
|
||||||
|
|
||||||
|
Accepted examples include a YAML library for configuration and the AWS SDK for S3-compatible storage.
|
||||||
|
|
||||||
|
### Explicit orchestration
|
||||||
|
|
||||||
|
The pipeline should remain stage-driven and explicit. New behavior should be added through clear stage, adapter, config, or manifest contracts rather than implicit side effects or generic workflow abstraction.
|
||||||
|
|
||||||
|
## Stage Design
|
||||||
|
|
||||||
|
Each stage should have a clear scope of responsibility.
|
||||||
|
|
||||||
|
A stage should define:
|
||||||
|
|
||||||
|
- its purpose;
|
||||||
|
- required input state;
|
||||||
|
- produced output state;
|
||||||
|
- config fields it consumes;
|
||||||
|
- external adapters it uses;
|
||||||
|
- manifest refs it reads or writes;
|
||||||
|
- skip, force, and resume behavior;
|
||||||
|
- failure behavior;
|
||||||
|
- tests that protect its contract.
|
||||||
|
|
||||||
|
Stages should avoid reaching across boundaries. If shared behavior is needed, prefer a helper or service with a narrow interface over duplicating ad hoc logic between stages.
|
||||||
|
|
||||||
|
## Transactionality and Resume
|
||||||
|
|
||||||
|
A stage should behave transactionally.
|
||||||
|
|
||||||
|
A stage is complete only when its outputs have been written, validated, and recorded in the manifest. If a stage fails, Narratio should preserve enough local state for inspection, recovery, and resume.
|
||||||
|
|
||||||
|
A failed or incomplete run must not be treated as successful. Later stages should depend on manifest-recorded success, not merely on incidental files existing on disk.
|
||||||
|
|
||||||
|
## Manifest Model
|
||||||
|
|
||||||
|
The manifest is the durable local ledger for a run.
|
||||||
|
|
||||||
|
It should record:
|
||||||
|
|
||||||
|
- run identity;
|
||||||
|
- stage status;
|
||||||
|
- input and output refs;
|
||||||
|
- logs and generated config refs;
|
||||||
|
- checksums or provenance where useful;
|
||||||
|
- non-secret adapter and archive metadata.
|
||||||
|
|
||||||
|
Resume behavior should be manifest-driven. Filesystem state may be inspected and validated, but it should not replace manifest stage state as the source of run progress.
|
||||||
|
|
||||||
|
## Adapter Boundaries
|
||||||
|
|
||||||
|
Adapters own external integration details.
|
||||||
|
|
||||||
|
Expected boundaries:
|
||||||
|
|
||||||
|
- WhisperX HTTP details stay in the WhisperX adapter.
|
||||||
|
- Seriatim CLI construction stays in the Seriatim adapter.
|
||||||
|
- Audita CLI construction stays in the Audita adapter.
|
||||||
|
- Scriptorium CLI construction stays in the Scriptorium adapter.
|
||||||
|
- Object-storage details stay behind the storage adapter interface.
|
||||||
|
- AWS SDK types stay inside the S3 storage implementation.
|
||||||
|
|
||||||
|
Stage code should express intent in Narratio terms and call adapters through narrow contracts.
|
||||||
|
|
||||||
|
## Configuration Philosophy
|
||||||
|
|
||||||
|
Configuration should be strict, explicit, and operator-friendly.
|
||||||
|
|
||||||
|
Principles:
|
||||||
|
|
||||||
|
- YAML decoding should reject unknown fields.
|
||||||
|
- Defaults should be centralized and testable.
|
||||||
|
- Empty configured values should not silently override meaningful defaults.
|
||||||
|
- Session templating should remain narrow and deterministic.
|
||||||
|
- Template support should serve operator convenience, not become a general configuration language.
|
||||||
|
|
||||||
|
Narratio should not become a secondary configuration system for downstream tools. Seriatim, Audita, and Scriptorium should own their runtime defaults wherever practical. Narratio should pass required stage-contract paths and explicit operator overrides.
|
||||||
|
|
||||||
|
## Path and Storage Discipline
|
||||||
|
|
||||||
|
Local and remote paths are part of Narratio’s application contract.
|
||||||
|
|
||||||
|
Code should use centralized path helpers for workspace, spool, session, run, artifact, log, config, and archive paths. Stages should avoid reconstructing canonical paths through scattered string concatenation.
|
||||||
|
|
||||||
|
Storage backends should receive explicit bucket-relative keys. Storage implementations should not infer campaign, session, run, or root-prefix semantics.
|
||||||
|
|
||||||
|
## Archive Invariants
|
||||||
|
|
||||||
|
Archive behavior must preserve a clear commit boundary.
|
||||||
|
|
||||||
|
A remote run is current only after the archive stage has successfully uploaded the run record, required promoted outputs, `current/manifest.json`, and finally `current/run_id.txt`.
|
||||||
|
|
||||||
|
`current/run_id.txt` is the final remote commit marker and must be written last.
|
||||||
|
|
||||||
|
Failed, incomplete, skipped, or uncommitted archive attempts must not be presented as current remote state. Local cleanup is permitted only after successful archive commit and only when explicitly configured.
|
||||||
|
|
||||||
|
## Security and Privacy
|
||||||
|
|
||||||
|
Narratio handles private campaign material.
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
|
||||||
|
- Do not store raw secrets in pipeline or session YAML.
|
||||||
|
- Use environment variable names or secret-file references for secret handling.
|
||||||
|
- Do not write raw secret values to manifests, logs, generated configs, or archive metadata.
|
||||||
|
- Treat transcripts, generated artifacts, prompts, reports, and logs as potentially sensitive.
|
||||||
|
- Avoid logging transcript or prompt content unless there is a deliberate diagnostic reason.
|
||||||
|
|
||||||
|
## Diagnostics
|
||||||
|
|
||||||
|
Diagnostics should be durable and discoverable, but distinct from canonical outputs.
|
||||||
|
|
||||||
|
Logs, reports, generated invocation/config files, and render-debug files support debugging. Transcript tiers and configured artifacts are pipeline products.
|
||||||
|
|
||||||
|
Manifest refs should preserve that distinction.
|
||||||
|
|
||||||
|
## Determinism
|
||||||
|
|
||||||
|
Where practical, Narratio should prefer deterministic behavior:
|
||||||
|
|
||||||
|
- stable local path layout;
|
||||||
|
- stable remote key layout;
|
||||||
|
- sorted upload order;
|
||||||
|
- predictable generated config files;
|
||||||
|
- repeatable command construction;
|
||||||
|
- tests that do not depend on live external services.
|
||||||
|
|
||||||
|
Run IDs and timestamps may be intentionally variable, but surrounding behavior should remain testable.
|
||||||
|
|
||||||
|
## Testing Expectations
|
||||||
|
|
||||||
|
Core behavior should be testable without live external services.
|
||||||
|
|
||||||
|
Tests should cover:
|
||||||
|
|
||||||
|
- config loading, defaults, and validation;
|
||||||
|
- CLI parsing and command construction;
|
||||||
|
- path helpers;
|
||||||
|
- manifest transitions;
|
||||||
|
- stage success, failure, skip, and resume behavior;
|
||||||
|
- adapter command construction;
|
||||||
|
- fake storage behavior;
|
||||||
|
- archive commit ordering;
|
||||||
|
- example config validity where practical.
|
||||||
|
|
||||||
|
Live S3, WhisperX, LLM, or subprocess integration tests should be explicit integration tests, not required for ordinary unit test runs.
|
||||||
|
|
||||||
|
## Documentation Expectations
|
||||||
|
|
||||||
|
Documentation must follow `docs/documentation/policy.md`.
|
||||||
|
|
||||||
|
Current behavior belongs in user-facing docs and `docs/internal/`. Future, planned, aspirational, experimental, or unimplemented work belongs only under `docs/roadmap/`.
|
||||||
|
|
||||||
|
`docs/architecture.md` should remain concise and principle-focused. It should not duplicate the full config reference, CLI reference, operations guide, or internal stage documentation.
|
||||||
|
|
||||||
|
## Non-Goals
|
||||||
|
|
||||||
|
Narratio is not:
|
||||||
|
|
||||||
|
- a generic DAG or workflow engine;
|
||||||
|
- a replacement configuration layer for Seriatim, Audita, or Scriptorium;
|
||||||
|
- a storage backend abstraction beyond the needs of this pipeline;
|
||||||
|
- a place to embed raw secrets;
|
||||||
|
- a place for stage logic to depend directly on AWS SDK types or downstream tool internals;
|
||||||
|
- a prompt-authoring system.
|
||||||
222
docs/cli.md
Normal file
222
docs/cli.md
Normal file
@@ -0,0 +1,222 @@
|
|||||||
|
# CLI
|
||||||
|
|
||||||
|
## Shortest Useful Command
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run --session-id 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
This command uses default config discovery for `pipeline.yml` and `session.yml`; both files must be discoverable unless you pass explicit `--config` and `--session` paths.
|
||||||
|
|
||||||
|
## Command Overview
|
||||||
|
|
||||||
|
Implemented commands:
|
||||||
|
|
||||||
|
- `run`: execute pipeline stages and persist manifest state.
|
||||||
|
- `plan`: validate config, prepare workspace layout, and print stage run/skip decisions.
|
||||||
|
- `resume`: continue from first non-succeeded stage unless forced.
|
||||||
|
- `status`: read and print stage statuses from an existing manifest.
|
||||||
|
- `run-stage`: execute exactly one stage.
|
||||||
|
|
||||||
|
Unknown commands print usage and exit non-zero.
|
||||||
|
|
||||||
|
For config semantics, see [docs/config.md](./config.md). For operator lifecycle and recovery, see [docs/operations.md](./operations.md).
|
||||||
|
|
||||||
|
## Complete Flag Reference
|
||||||
|
|
||||||
|
### `run`
|
||||||
|
|
||||||
|
- `--config <path>`: optional explicit `pipeline.yml` path.
|
||||||
|
- `--session <path>`: optional explicit `session.yml` path.
|
||||||
|
- `--session-id <value>`: session template variable value.
|
||||||
|
- `--force`: force stage execution.
|
||||||
|
- `--artifacts <names>`: analyze artifact keys to execute (repeatable or comma-separated).
|
||||||
|
|
||||||
|
### `plan`
|
||||||
|
|
||||||
|
- `--config <path>`
|
||||||
|
- `--session <path>`
|
||||||
|
- `--session-id <value>`
|
||||||
|
- `--force`
|
||||||
|
|
||||||
|
### `resume`
|
||||||
|
|
||||||
|
- `--config <path>`
|
||||||
|
- `--session <path>`
|
||||||
|
- `--session-id <value>`
|
||||||
|
- `--force`
|
||||||
|
- `--artifacts <names>`: analyze artifact keys to execute (repeatable or comma-separated).
|
||||||
|
|
||||||
|
### `run-stage`
|
||||||
|
|
||||||
|
- `--config <path>`
|
||||||
|
- `--session <path>`
|
||||||
|
- `--session-id <value>`
|
||||||
|
- `--force`
|
||||||
|
- `--artifacts <names>`: analyze artifact keys to execute (repeatable or comma-separated).
|
||||||
|
- positional `<stage>`: required stage name.
|
||||||
|
|
||||||
|
Valid stage names:
|
||||||
|
|
||||||
|
- `prepare`
|
||||||
|
- `transcribe`
|
||||||
|
- `merge`
|
||||||
|
- `polish`
|
||||||
|
- `normalize`
|
||||||
|
- `trim`
|
||||||
|
- `analyze`
|
||||||
|
- `archive`
|
||||||
|
- `notify`
|
||||||
|
|
||||||
|
### `status`
|
||||||
|
|
||||||
|
- `--manifest <path>`: required manifest path.
|
||||||
|
|
||||||
|
## Command Reference
|
||||||
|
|
||||||
|
### `run`
|
||||||
|
|
||||||
|
Purpose:
|
||||||
|
- Execute configured stages in canonical order.
|
||||||
|
|
||||||
|
Syntax:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force] [--artifacts <name[,name...]>]
|
||||||
|
```
|
||||||
|
|
||||||
|
Success output:
|
||||||
|
- `narratio run: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
|
||||||
|
|
||||||
|
Common failure cases:
|
||||||
|
- missing default config/session paths when flags omitted.
|
||||||
|
- invalid template/rendered session mismatch.
|
||||||
|
- unknown/invalid `--artifacts` value.
|
||||||
|
- `--artifacts` with unknown configured artifact key.
|
||||||
|
|
||||||
|
### `plan`
|
||||||
|
|
||||||
|
Purpose:
|
||||||
|
- Validate config, load secrets (if configured), prepare workdir, and print stage run/skip decisions.
|
||||||
|
|
||||||
|
Syntax:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio plan [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force]
|
||||||
|
```
|
||||||
|
|
||||||
|
Success output includes:
|
||||||
|
- `narratio plan: workdir prepared at <path>`
|
||||||
|
- one line per stage (`<stage>: run|skip`)
|
||||||
|
- `totals: run=<n> skip=<n>`
|
||||||
|
|
||||||
|
Common failure cases:
|
||||||
|
- same config/session discovery and validation failures as `run`.
|
||||||
|
- secrets directory read failures when `pipeline.secrets.env_dir` is configured.
|
||||||
|
|
||||||
|
### `resume`
|
||||||
|
|
||||||
|
Purpose:
|
||||||
|
- Continue from session-manifest stage status.
|
||||||
|
|
||||||
|
Syntax:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio resume [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force] [--artifacts <name[,name...]>]
|
||||||
|
```
|
||||||
|
|
||||||
|
Success output:
|
||||||
|
- `narratio resume: session <session_id> has no remaining stages`
|
||||||
|
- or `narratio resume: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
|
||||||
|
|
||||||
|
Common failure cases:
|
||||||
|
- same discovery/template/validation failures as `run`.
|
||||||
|
- manifest load errors when existing manifest is unreadable.
|
||||||
|
- invalid or unknown artifact selections.
|
||||||
|
|
||||||
|
### `status`
|
||||||
|
|
||||||
|
Purpose:
|
||||||
|
- Inspect one manifest file without executing stages.
|
||||||
|
|
||||||
|
Syntax:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio status --manifest <manifest.json>
|
||||||
|
```
|
||||||
|
|
||||||
|
Success output includes:
|
||||||
|
- `session_id: <id>`
|
||||||
|
- `updated_at: <timestamp>`
|
||||||
|
- `stages:` entries (`- <stage>: <status>`)
|
||||||
|
|
||||||
|
Common failure cases:
|
||||||
|
- missing `--manifest`.
|
||||||
|
- unreadable or invalid manifest path.
|
||||||
|
|
||||||
|
### `run-stage`
|
||||||
|
|
||||||
|
Purpose:
|
||||||
|
- Execute exactly one stage.
|
||||||
|
|
||||||
|
Syntax:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run-stage [--config <pipeline.yml>] [--session <session.yml>] [--session-id <id>] [--force] [--artifacts <name[,name...]>] <stage>
|
||||||
|
```
|
||||||
|
|
||||||
|
Success output:
|
||||||
|
- `narratio run-stage: stage=<name> executed=<n> skipped=<n> force=<true|false>; manifest=<path>`
|
||||||
|
|
||||||
|
`--artifacts` behavior:
|
||||||
|
- accepted only when `<stage>` is `analyze`.
|
||||||
|
- names are normalized (trimmed, deduplicated, sorted).
|
||||||
|
- unknown configured artifact keys fail.
|
||||||
|
|
||||||
|
Common failure cases:
|
||||||
|
- missing stage positional arg.
|
||||||
|
- unknown stage name.
|
||||||
|
- using `--artifacts` with any non-`analyze` stage.
|
||||||
|
|
||||||
|
## Common Workflows
|
||||||
|
|
||||||
|
Default-discovery run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run --session-id 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
Run only selected analyze artifacts:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run --session-id 2026-04-04 --artifacts session_recap,player_handout
|
||||||
|
```
|
||||||
|
|
||||||
|
Resume with selected analyze artifacts:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio resume --session-id 2026-04-04 --artifacts player_handout
|
||||||
|
```
|
||||||
|
|
||||||
|
Run only analyze stage with selected artifacts:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run-stage --session-id 2026-04-04 --artifacts player_handout analyze
|
||||||
|
```
|
||||||
|
|
||||||
|
## Diagnostic / Recovery Commands
|
||||||
|
|
||||||
|
Inspect stage status:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio status --manifest <manifest.json>
|
||||||
|
```
|
||||||
|
|
||||||
|
Get manifest path from previous output:
|
||||||
|
- `run`, `resume`, and `run-stage` print `manifest=<path>` on success.
|
||||||
|
|
||||||
|
## `--artifacts` and `--force`
|
||||||
|
|
||||||
|
- `--artifacts` filters which configured artifacts are executable when analyze runs.
|
||||||
|
- `--artifacts` does not imply `--force`.
|
||||||
|
- If analyze is already `succeeded` and `--force` is not set, runner-level skip still applies.
|
||||||
304
docs/config.md
Normal file
304
docs/config.md
Normal file
@@ -0,0 +1,304 @@
|
|||||||
|
# Configuration
|
||||||
|
|
||||||
|
## 1. Overview
|
||||||
|
|
||||||
|
Narratio loads two YAML files:
|
||||||
|
|
||||||
|
- `pipeline.yml`: pipeline-level runtime configuration.
|
||||||
|
- `session.yml`: per-session metadata and input selection.
|
||||||
|
|
||||||
|
These commands load and validate both files before running:
|
||||||
|
|
||||||
|
- `narratio run`
|
||||||
|
- `narratio plan`
|
||||||
|
- `narratio resume`
|
||||||
|
- `narratio run-stage`
|
||||||
|
|
||||||
|
Behavior:
|
||||||
|
|
||||||
|
- strict YAML decode is enabled (`KnownFields(true)`): unknown fields fail.
|
||||||
|
- session templates render before session YAML decode.
|
||||||
|
- defaults are applied for optional pipeline fields.
|
||||||
|
- validation enforces required fields, value formats, and cross-field constraints.
|
||||||
|
|
||||||
|
## 2. Config file discovery
|
||||||
|
|
||||||
|
Pipeline config lookup for `run`, `plan`, `resume`, and `run-stage`:
|
||||||
|
|
||||||
|
- If `--config <path>` is provided, that path is used.
|
||||||
|
- If omitted, Narratio searches in order:
|
||||||
|
1. `/usr/local/etc/narratio/pipeline.yml`
|
||||||
|
2. `/etc/narratio/pipeline.yml`
|
||||||
|
- First existing file wins.
|
||||||
|
|
||||||
|
## 3. Session file discovery and templating
|
||||||
|
|
||||||
|
Session config lookup for `run`, `plan`, `resume`, and `run-stage`:
|
||||||
|
|
||||||
|
- If `--session <path>` is provided, that path is used.
|
||||||
|
- If omitted, Narratio searches in order:
|
||||||
|
1. `./session.yml`
|
||||||
|
2. `/usr/local/etc/narratio/session.yml`
|
||||||
|
3. `/etc/narratio/session.yml`
|
||||||
|
- First existing file wins.
|
||||||
|
|
||||||
|
Template behavior:
|
||||||
|
|
||||||
|
- Supported placeholders:
|
||||||
|
- `{{session_id}}`
|
||||||
|
- `{{ session_id }}`
|
||||||
|
- `--session-id <value>` supplies the placeholder value.
|
||||||
|
- unresolved placeholders fail load.
|
||||||
|
- if rendered `session_id` mismatches `--session-id`, load fails.
|
||||||
|
|
||||||
|
## 4. Minimal pipeline config
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
whisperx:
|
||||||
|
transcribe_url: "https://transcription.example.com/transcribe"
|
||||||
|
```
|
||||||
|
|
||||||
|
Why this is sufficient:
|
||||||
|
|
||||||
|
- `whisperx.transcribe_url` is required.
|
||||||
|
- `workspace.root` defaults to `/var/lib/narratio`.
|
||||||
|
- optional sections (`seriatim`, `audita`, `archive`, `scriptorium`, `trim`, `normalize`, etc.) receive defaults or stay inactive.
|
||||||
|
|
||||||
|
## 5. Minimal session template
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
session_id: "{{ session_id }}"
|
||||||
|
campaign: sample-campaign
|
||||||
|
inputs:
|
||||||
|
audio_dir: ./audio
|
||||||
|
speakers_file: ./examples/speakers.yml
|
||||||
|
autocorrect_file: ./examples/autocorrect.yml
|
||||||
|
glossary_file: ./examples/glossary.yml
|
||||||
|
```
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run --config /path/to/pipeline.yml --session ./session.yml --session-id 2026-05-03
|
||||||
|
```
|
||||||
|
|
||||||
|
## 6. Production-oriented config
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
workspace:
|
||||||
|
root: /var/lib/narratio/workspace
|
||||||
|
cleanup_after_archive: true
|
||||||
|
|
||||||
|
storage:
|
||||||
|
backend: s3
|
||||||
|
s3:
|
||||||
|
bucket: my-dnd-archive
|
||||||
|
root_prefix: dnd
|
||||||
|
region: us-east-1
|
||||||
|
access_key_id_env: OBJECT_STORAGE_KEY_ID
|
||||||
|
secret_access_key_env: OBJECT_STORAGE_KEY
|
||||||
|
|
||||||
|
spool:
|
||||||
|
root: /var/spool/narratio
|
||||||
|
delete_audio_after_archive: true
|
||||||
|
|
||||||
|
archive:
|
||||||
|
enabled: true
|
||||||
|
upload_run: true
|
||||||
|
promote_artifacts:
|
||||||
|
- from: transcripts/trimmed.json
|
||||||
|
to: transcripts/trimmed.json
|
||||||
|
required: true
|
||||||
|
- from: artifacts/session_recap.md
|
||||||
|
to: artifacts/session_recap.md
|
||||||
|
required: true
|
||||||
|
|
||||||
|
whisperx:
|
||||||
|
transcribe_url: "https://transcription.example.com/transcribe"
|
||||||
|
|
||||||
|
scriptorium:
|
||||||
|
artifacts:
|
||||||
|
session_recap:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd.session_recap
|
||||||
|
output_path: artifacts/session_recap.md
|
||||||
|
inputs:
|
||||||
|
transcript:
|
||||||
|
source: narratio.transcript.trimmed
|
||||||
|
required: true
|
||||||
|
```
|
||||||
|
|
||||||
|
Operational notes:
|
||||||
|
|
||||||
|
- archive promotion is explicit and path-based via `archive.promote_artifacts`.
|
||||||
|
- Narratio does not auto-promote all generated analyze artifacts.
|
||||||
|
|
||||||
|
## 7. Full pipeline reference
|
||||||
|
|
||||||
|
| Path | Type | Required | Default |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| `pipeline.workspace.root` | string | No | `/var/lib/narratio` |
|
||||||
|
| `pipeline.workspace.cleanup_after_archive` | bool | No | `false` |
|
||||||
|
| `pipeline.secrets.env_dir` | string | Conditional | none |
|
||||||
|
| `pipeline.storage.backend` | string | No | empty |
|
||||||
|
| `pipeline.storage.bucket` | string | No | empty |
|
||||||
|
| `pipeline.storage.prefix` | string | No | empty |
|
||||||
|
| `pipeline.storage.s3.bucket` | string | Conditional | empty |
|
||||||
|
| `pipeline.storage.s3.root_prefix` | string | No | `dnd` |
|
||||||
|
| `pipeline.storage.s3.region` | string | No | empty |
|
||||||
|
| `pipeline.storage.s3.endpoint` | string | No | empty |
|
||||||
|
| `pipeline.storage.s3.force_path_style` | bool | No | `false` |
|
||||||
|
| `pipeline.storage.s3.access_key_id_env` | string | No | `OBJECT_STORAGE_KEY_ID` |
|
||||||
|
| `pipeline.storage.s3.secret_access_key_env` | string | No | `OBJECT_STORAGE_KEY` |
|
||||||
|
| `pipeline.spool.root` | string | No | `/var/spool/narratio` |
|
||||||
|
| `pipeline.spool.delete_audio_after_archive` | bool | No | `false` |
|
||||||
|
| `pipeline.archive.enabled` | bool | No | `true` |
|
||||||
|
| `pipeline.archive.upload_run` | bool | No | `true` |
|
||||||
|
| `pipeline.archive.promote_artifacts[]` | list | No | trimmed + session_recap rules |
|
||||||
|
| `pipeline.archive.promote_artifacts[].from` | string | Yes (per rule) | none |
|
||||||
|
| `pipeline.archive.promote_artifacts[].to` | string | Yes (per rule) | none |
|
||||||
|
| `pipeline.archive.promote_artifacts[].required` | bool | No | `true` |
|
||||||
|
| `pipeline.whisperx.transcribe_url` | string | Yes | none |
|
||||||
|
| `pipeline.whisperx.language` | string | No | `en` |
|
||||||
|
| `pipeline.whisperx.timeout` | duration string | No | `30m` |
|
||||||
|
| `pipeline.whisperx.retries` | int | No | `3` |
|
||||||
|
| `pipeline.whisperx.retry_delay` | duration string | No | `2s` |
|
||||||
|
| `pipeline.whisperx.concurrency` | int | No | `2` |
|
||||||
|
| `pipeline.seriatim.binary` | string | No | `seriatim` |
|
||||||
|
| `pipeline.seriatim.timeout` | duration string | No | `10m` |
|
||||||
|
| `pipeline.seriatim.output_schema` | string | No | `seriatim-intermediate` |
|
||||||
|
| `pipeline.seriatim.coalesce_gap` | float | No | `3.0` |
|
||||||
|
| `pipeline.seriatim.report` | bool | No | `true` |
|
||||||
|
| `pipeline.seriatim.env.overlap_word_run_gap` | float | No | unset |
|
||||||
|
| `pipeline.seriatim.env.overlap_word_run_reorder_window` | float | No | unset |
|
||||||
|
| `pipeline.seriatim.env.backchannel_max_duration` | float | No | unset |
|
||||||
|
| `pipeline.seriatim.env.filler_max_duration` | float | No | unset |
|
||||||
|
| `pipeline.audita.binary` | string | No | `audita` |
|
||||||
|
| `pipeline.audita.timeout` | duration string | No | `3h` |
|
||||||
|
| `pipeline.audita.llm_api_key_env` | string | No | empty |
|
||||||
|
| `pipeline.audita.modules[]` | list[string] | No | empty |
|
||||||
|
| `pipeline.audita.base_url` | string | No | empty |
|
||||||
|
| `pipeline.audita.model` | string | No | empty |
|
||||||
|
| `pipeline.audita.total_llm_concurrency` | int | No | unset |
|
||||||
|
| `pipeline.audita.proposal_llm_concurrency` | int | No | unset |
|
||||||
|
| `pipeline.audita.validation_model` | string | No | empty |
|
||||||
|
| `pipeline.audita.validation_llm_concurrency` | int | No | unset |
|
||||||
|
| `pipeline.audita.transcript_description` | string | No | empty |
|
||||||
|
| `pipeline.audita.config_path` | string | No | empty |
|
||||||
|
| `pipeline.audita.output_schema` | string | No | empty |
|
||||||
|
| `pipeline.audita.work_dir_retention` | string | No | empty |
|
||||||
|
| `pipeline.audita.report` | bool | No | `true` |
|
||||||
|
| `pipeline.normalize.output_path` | string | No | `transcripts/normalized.json` |
|
||||||
|
| `pipeline.normalize.output_schema` | string | No | `seriatim-intermediate` |
|
||||||
|
| `pipeline.normalize.report` | bool | No | `true` |
|
||||||
|
| `pipeline.trim.enabled` | bool | No | `false` |
|
||||||
|
| `pipeline.trim.output_path` | string | Conditional | none |
|
||||||
|
| `pipeline.trim.bounds.prompt_id` | string | Conditional | none |
|
||||||
|
| `pipeline.trim.bounds.profile_id` | string | No | empty |
|
||||||
|
| `pipeline.trim.bounds.transcript_input_name` | string | Conditional | none |
|
||||||
|
| `pipeline.trim.bounds.output_path` | string | Conditional | none |
|
||||||
|
| `pipeline.trim.bounds.timeout` | duration string | No | `10m` |
|
||||||
|
| `pipeline.trim.bounds.render_debug` | bool | No | `false` |
|
||||||
|
| `pipeline.trim.bounds.render_output_path` | string | Conditional | none |
|
||||||
|
| `pipeline.trim.seriatim.report` | bool | No | `false` |
|
||||||
|
| `pipeline.scriptorium.binary` | string | No | `scriptorium` |
|
||||||
|
| `pipeline.scriptorium.config_path` | string | No | empty |
|
||||||
|
| `pipeline.scriptorium.timeout` | duration string | No | `10m` |
|
||||||
|
| `pipeline.scriptorium.render_debug` | bool | No | `false` |
|
||||||
|
| `pipeline.scriptorium.artifacts` | map | No | empty |
|
||||||
|
| `pipeline.scriptorium.artifacts.<name>.enabled` | bool | No | `false` |
|
||||||
|
| `pipeline.scriptorium.artifacts.<name>.depends_on[]` | list[string] | No | empty |
|
||||||
|
| `pipeline.scriptorium.artifacts.<name>.render_debug` | bool | No | unset |
|
||||||
|
| `pipeline.scriptorium.artifacts.<name>.prompt_id` | string | Conditional | none |
|
||||||
|
| `pipeline.scriptorium.artifacts.<name>.profile_id` | string | No | empty |
|
||||||
|
| `pipeline.scriptorium.artifacts.<name>.output_path` | string | Conditional | none |
|
||||||
|
| `pipeline.scriptorium.artifacts.<name>.timeout` | duration string | No | empty |
|
||||||
|
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source` | string | Conditional | none |
|
||||||
|
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.artifact` | string | No | empty |
|
||||||
|
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.path` | string | No | empty |
|
||||||
|
| `pipeline.scriptorium.artifacts.<name>.inputs.<key>.required` | bool | No | `false` |
|
||||||
|
| `pipeline.scriptorium.artifacts.<name>.vars.<key>` | map value | No | empty |
|
||||||
|
| `pipeline.analyzer.binary_path` | string | No | empty |
|
||||||
|
| `pipeline.analyzer.timeout` | duration string | No | empty |
|
||||||
|
| `pipeline.analyzer.artifacts.output_dir` | string | No | empty |
|
||||||
|
| `pipeline.analyzer.artifacts.types[]` | list[string] | No | empty |
|
||||||
|
| `pipeline.notification.backend` | string | No | empty |
|
||||||
|
| `pipeline.notification.recipient` | string | No | empty |
|
||||||
|
| `pipeline.notification.timeout` | duration string | No | empty |
|
||||||
|
|
||||||
|
Scriptorium artifact-key and dependency rules:
|
||||||
|
|
||||||
|
- artifact keys must match `^[a-z][a-z0-9_]*$`.
|
||||||
|
- enabled artifacts require `prompt_id` and `output_path`.
|
||||||
|
- `output_path` must be relative, traversal-safe, and under `artifacts/`.
|
||||||
|
- configured artifact input sources use `narratio.artifact.<name>`.
|
||||||
|
- if input source references `narratio.artifact.<name>`, artifact `<name>` must exist and must be listed in `depends_on`.
|
||||||
|
- every `depends_on` entry must be a configured artifact key.
|
||||||
|
- self-dependency is rejected.
|
||||||
|
- enabled dependency cycles are rejected.
|
||||||
|
- any artifact referenced by `depends_on` or `narratio.artifact.<name>` source must define `output_path` (even if not enabled).
|
||||||
|
|
||||||
|
Allowed `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source` values:
|
||||||
|
|
||||||
|
- `previous_session_artifact`
|
||||||
|
- `narratio.transcript.merged`
|
||||||
|
- `narratio.transcript.polished`
|
||||||
|
- `narratio.transcript.full`
|
||||||
|
- `narratio.transcript.trimmed`
|
||||||
|
- `narratio.bounds.session`
|
||||||
|
- `narratio.artifact.<configured_artifact_key>`
|
||||||
|
|
||||||
|
## 8. Full session reference
|
||||||
|
|
||||||
|
| Path | Type | Required | Default |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| `session.session_id` | string | Yes | none |
|
||||||
|
| `session.campaign` | string | Yes | none |
|
||||||
|
| `session.date` | string | No | empty |
|
||||||
|
| `session.title` | string | No | empty |
|
||||||
|
| `session.inputs.audio_dir` | string | Conditional | empty |
|
||||||
|
| `session.inputs.audio_files[]` | list[string] | Conditional | empty |
|
||||||
|
| `session.inputs.audio_s3.prefix` | string | Conditional | none |
|
||||||
|
| `session.inputs.speakers_file` | string | Yes | none |
|
||||||
|
| `session.inputs.autocorrect_file` | string | Yes | none |
|
||||||
|
| `session.inputs.glossary_file` | string | Yes | none |
|
||||||
|
|
||||||
|
Audio-source rule:
|
||||||
|
|
||||||
|
- configure exactly one mode:
|
||||||
|
- `audio_dir`, or
|
||||||
|
- `audio_files` (at least one), or
|
||||||
|
- `audio_s3.prefix`
|
||||||
|
- `audio_s3` cannot be combined with local audio fields.
|
||||||
|
|
||||||
|
## 9. Secrets
|
||||||
|
|
||||||
|
Narratio supports filesystem-based secret injection via `pipeline.secrets.env_dir`.
|
||||||
|
|
||||||
|
Behavior:
|
||||||
|
|
||||||
|
- `env_dir` may be absolute or relative.
|
||||||
|
- relative `env_dir` resolves from current working directory.
|
||||||
|
- files with valid env-var names (`[A-Za-z_][A-Za-z0-9_]*`) are loaded.
|
||||||
|
- values are loaded from file contents with trailing newline trimming.
|
||||||
|
- existing process env vars are preserved.
|
||||||
|
- invalid names and subdirectories are skipped.
|
||||||
|
- missing/unreadable `env_dir` fails command execution.
|
||||||
|
|
||||||
|
Guidance:
|
||||||
|
|
||||||
|
- do not put secret values directly in YAML.
|
||||||
|
- configure env var names in config and provide values via env/secrets files.
|
||||||
|
|
||||||
|
## 10. Examples
|
||||||
|
|
||||||
|
Maintained examples:
|
||||||
|
|
||||||
|
- `examples/pipeline.minimal.yml`
|
||||||
|
- `examples/pipeline.production.yml`
|
||||||
|
- `examples/pipeline.full.annotated.yml`
|
||||||
|
- `examples/session.template.yml`
|
||||||
|
- `examples/session.local-audio.yml`
|
||||||
|
- `examples/session.s3-audio.yml`
|
||||||
|
|
||||||
|
These examples are validated by `internal/config` tests.
|
||||||
92
docs/development.md
Normal file
92
docs/development.md
Normal file
@@ -0,0 +1,92 @@
|
|||||||
|
# Development Guide
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Canonical contributor workflow and engineering conventions for implemented Narratio behavior.
|
||||||
|
|
||||||
|
## Repository layout
|
||||||
|
|
||||||
|
- `cmd/narratio/`: CLI entrypoint.
|
||||||
|
- `internal/app/`: command handlers, plan/run/resume orchestration, cleanup gates, secrets loading.
|
||||||
|
- `internal/config/`: strict YAML loading, defaults, and validation.
|
||||||
|
- `internal/stage/`: stage implementations and stage registry/order.
|
||||||
|
- `internal/adapters/`: external boundary adapters (WhisperX, Seriatim, Audita, Scriptorium, storage, notify).
|
||||||
|
- `internal/manifest/`: session/run manifest types and persistence.
|
||||||
|
- `internal/artifacts/`: canonical local/remote path helpers and local artifact store.
|
||||||
|
- `docs/`: canonical documentation set.
|
||||||
|
- `examples/`: maintained config examples used by tests.
|
||||||
|
|
||||||
|
## Build and test commands
|
||||||
|
|
||||||
|
- Run focused CLI behavior checks:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
go test ./internal/app -run TestExecute -v
|
||||||
|
```
|
||||||
|
|
||||||
|
- Run config example load/validate checks:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
go test ./internal/config -run TestExamplesLoadAndValidate -v
|
||||||
|
```
|
||||||
|
|
||||||
|
- Run full test suite:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
go test ./...
|
||||||
|
```
|
||||||
|
|
||||||
|
## Coding conventions
|
||||||
|
|
||||||
|
- Keep orchestration explicit and stage-driven; do not introduce generic workflow/DAG abstractions.
|
||||||
|
- Keep external-system details inside adapter packages; stages should consume Narratio-level contracts only.
|
||||||
|
- Use centralized path helpers from `internal/artifacts` rather than ad hoc path concatenation.
|
||||||
|
- Preserve manifest-driven state transitions (`running`, `succeeded`, `failed`, `skipped`, `stale`) as the source of run progress.
|
||||||
|
- Keep user/operator docs implementation-accurate; planned work belongs only under `docs/roadmap/`.
|
||||||
|
|
||||||
|
For design principles and invariants, see [docs/architecture.md](./architecture.md). For stage/adapter contracts, see [docs/internal/README.md](./internal/README.md).
|
||||||
|
|
||||||
|
## Dependency policy
|
||||||
|
|
||||||
|
- Prefer Go standard library where practical.
|
||||||
|
- Add third-party dependencies only when they provide clear value for required behavior.
|
||||||
|
- Keep dependency additions narrow to the boundary package that needs them.
|
||||||
|
|
||||||
|
## Change playbooks
|
||||||
|
|
||||||
|
### Add config fields
|
||||||
|
|
||||||
|
1. Add fields to config structs in `internal/config`.
|
||||||
|
2. Set defaults in `internal/config/defaults.go` when appropriate.
|
||||||
|
3. Add validation rules in `internal/config/validate.go`.
|
||||||
|
4. Add or update load/validate tests in `internal/config/*_test.go`.
|
||||||
|
5. Update canonical config docs and examples:
|
||||||
|
- [docs/config.md](./config.md)
|
||||||
|
- relevant files under `examples/`
|
||||||
|
|
||||||
|
### Add CLI flags or commands
|
||||||
|
|
||||||
|
1. Update command parsing and behavior in `internal/app`.
|
||||||
|
2. Add or update command tests (`TestExecute` and command-specific tests).
|
||||||
|
3. Update [docs/cli.md](./cli.md) and, if operator workflow changes, [docs/operations.md](./operations.md).
|
||||||
|
|
||||||
|
### Add or modify stages/adapters
|
||||||
|
|
||||||
|
1. Implement stage behavior in `internal/stage` with clear input/output boundaries.
|
||||||
|
2. Keep external transport/subprocess details in `internal/adapters`.
|
||||||
|
3. Preserve manifest and promotion semantics expected by runner and archive logic.
|
||||||
|
4. Add/update stage and adapter tests.
|
||||||
|
5. Update internal component contracts in `docs/internal/`.
|
||||||
|
|
||||||
|
### Update examples
|
||||||
|
|
||||||
|
1. Keep canonical examples only in `examples/`.
|
||||||
|
2. Ensure examples load and validate through runtime config paths.
|
||||||
|
3. Update `internal/config/load_validate_test.go` as needed.
|
||||||
|
4. Update links in `docs/config.md` if example filenames change.
|
||||||
|
|
||||||
|
### Update docs and roadmap
|
||||||
|
|
||||||
|
1. Keep implemented behavior in canonical docs (`README`, `docs/*.md`, `docs/internal/`).
|
||||||
|
2. Keep planned/unimplemented behavior only in `docs/roadmap/`.
|
||||||
|
3. After completing roadmap items, remove or mark them complete in `docs/roadmap/documentation.md`.
|
||||||
|
4. Run a link/path sweep before finalizing changes.
|
||||||
@@ -1,130 +0,0 @@
|
|||||||
# Workspace Architecture Implementation Plan (Status)
|
|
||||||
|
|
||||||
This document tracks the implemented workspace architecture and remaining work for v1.0.
|
|
||||||
|
|
||||||
## Current Architecture (Implemented)
|
|
||||||
|
|
||||||
Narratio now uses a canonical campaign-aware local layout:
|
|
||||||
|
|
||||||
```text
|
|
||||||
{workspace.root}/work/{campaign_id}/{session_id}/
|
|
||||||
manifest.json
|
|
||||||
current/
|
|
||||||
manifest.json
|
|
||||||
run_id.txt
|
|
||||||
inputs/
|
|
||||||
transcripts/
|
|
||||||
artifacts/
|
|
||||||
reports/
|
|
||||||
logs/
|
|
||||||
config/
|
|
||||||
runs/
|
|
||||||
{run_id}/
|
|
||||||
manifest.json
|
|
||||||
{stage}/
|
|
||||||
outputs/
|
|
||||||
logs/
|
|
||||||
reports/
|
|
||||||
config/
|
|
||||||
scratch/
|
|
||||||
```
|
|
||||||
|
|
||||||
Core behavior:
|
|
||||||
|
|
||||||
- Session manifest remains the skip/resume source of truth.
|
|
||||||
- Each invocation creates a run manifest at `runs/{run_id}/manifest.json`.
|
|
||||||
- Stage execution writes run-local artifacts and promotes durable outputs to canonical session paths.
|
|
||||||
- Archive uploads run records under `runs/{run_id}/`, applies promotion rules, then publishes `current/manifest.json` and `current/run_id.txt`.
|
|
||||||
- Analyze input resolution uses centralized artifact IDs with alias support.
|
|
||||||
- Forced upstream reruns mark downstream succeeded stages `stale` so later runs do not skip stale outputs.
|
|
||||||
|
|
||||||
## Section 4 Sequence Status
|
|
||||||
|
|
||||||
### Step 1: Campaign-aware session path model
|
|
||||||
|
|
||||||
Status: complete.
|
|
||||||
|
|
||||||
Implemented:
|
|
||||||
|
|
||||||
- Campaign-aware session and run path helpers.
|
|
||||||
- Campaign-aware artifact-store layout APIs.
|
|
||||||
- Canonical session manifest pathing under `work/{campaign}/{session}`.
|
|
||||||
|
|
||||||
### Step 2: Session manifest + run manifest scaffolding
|
|
||||||
|
|
||||||
Status: complete.
|
|
||||||
|
|
||||||
Implemented:
|
|
||||||
|
|
||||||
- Invocation-scoped run manifest type and store methods.
|
|
||||||
- Runner creates/saves run manifests per invocation.
|
|
||||||
- Session manifest remains authoritative for idempotent stage skipping.
|
|
||||||
|
|
||||||
### Step 3: Run-local stage execution + promotion
|
|
||||||
|
|
||||||
Status: complete.
|
|
||||||
|
|
||||||
Implemented:
|
|
||||||
|
|
||||||
- Run-local stage directory layout under `runs/{run_id}/{stage}`.
|
|
||||||
- Shared helpers for run-local output mapping and promotion to canonical durable paths.
|
|
||||||
- Producer run provenance recorded on durable artifact outputs.
|
|
||||||
|
|
||||||
### Step 4: Archive alignment
|
|
||||||
|
|
||||||
Status: complete.
|
|
||||||
|
|
||||||
Implemented:
|
|
||||||
|
|
||||||
- Canonical run-root/session-root resolution.
|
|
||||||
- Deterministic run-file collection and promotion source resolution.
|
|
||||||
- Current-pointer publication ordering retained (`current/manifest.json` then `current/run_id.txt`).
|
|
||||||
|
|
||||||
### Step 5: Artifact registry/resolver (analyze first consumer)
|
|
||||||
|
|
||||||
Status: complete.
|
|
||||||
|
|
||||||
Implemented:
|
|
||||||
|
|
||||||
- Central artifact resolver with canonical IDs:
|
|
||||||
- `narratio.transcript.merged`
|
|
||||||
- `narratio.transcript.polished`
|
|
||||||
- `narratio.transcript.full`
|
|
||||||
- `narratio.transcript.trimmed`
|
|
||||||
- `narratio.bounds.session`
|
|
||||||
- `narratio.artifact.session_recap`
|
|
||||||
- Backward-compatible aliases:
|
|
||||||
- `processed_transcript`
|
|
||||||
- `normalized_transcript`
|
|
||||||
- `trimmed_transcript`
|
|
||||||
- Analyze stage switched to resolver-based source resolution.
|
|
||||||
|
|
||||||
### Step 6: Minimal downstream invalidation for forced reruns
|
|
||||||
|
|
||||||
Status: complete.
|
|
||||||
|
|
||||||
Implemented:
|
|
||||||
|
|
||||||
- Deterministic downstream invalidation based on canonical stage order.
|
|
||||||
- On forced successful rerun of stage `X`, downstream succeeded stages are marked `stale`.
|
|
||||||
- Resume and non-forced runs naturally re-execute stale stages.
|
|
||||||
|
|
||||||
### Step 7: Legacy layout migration strategy
|
|
||||||
|
|
||||||
Status: intentionally skipped.
|
|
||||||
|
|
||||||
Decision:
|
|
||||||
|
|
||||||
- Automatic migration and legacy fallback compatibility are intentionally not implemented.
|
|
||||||
- The codebase targets canonical-only local layout behavior.
|
|
||||||
- Legacy local workspace state, if present, should be recreated or migrated manually outside Narratio.
|
|
||||||
|
|
||||||
## Remaining Work (v1.0)
|
|
||||||
|
|
||||||
No required workspace/run-history migration steps remain from Section 4.
|
|
||||||
|
|
||||||
Possible future enhancements (non-blocking):
|
|
||||||
|
|
||||||
- Full checksum/input-graph stale detection.
|
|
||||||
- Optional retention-policy expansion for run-history cleanup.
|
|
||||||
- Broader artifact-resolver adoption across additional stage consumers.
|
|
||||||
@@ -1,744 +0,0 @@
|
|||||||
# Narratio Workspace, Run History, and Artifact Resolution Architecture
|
|
||||||
|
|
||||||
## 1. Purpose
|
|
||||||
|
|
||||||
This document defines the intended v1.0 architecture for Narratio's local workspace layout, run history model, durable session outputs, manifest responsibilities, and artifact resolution contract.
|
|
||||||
|
|
||||||
Narratio is an idempotent session orchestrator. The command:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
narratio run --session-id 2026-05-07
|
|
||||||
```
|
|
||||||
|
|
||||||
means "bring the identified session to its desired completed state." It does **not** mean "always create an entirely new independent output tree and ignore prior session state."
|
|
||||||
|
|
||||||
This distinction drives the architecture:
|
|
||||||
|
|
||||||
* A **session** is the durable domain object and idempotency boundary.
|
|
||||||
* A **run** is an execution attempt that may update the session's durable state.
|
|
||||||
* Durable outputs live at the session level.
|
|
||||||
* Run-specific outputs, logs, generated configs, scratch files, and diagnostics live under `runs/{run_id}/`.
|
|
||||||
* Successful stage outputs are promoted from run-local locations into canonical session-level locations.
|
|
||||||
* The session manifest records current durable state.
|
|
||||||
* Run manifests record execution history and debugging/provenance details.
|
|
||||||
|
|
||||||
This model intentionally mirrors the S3 archive model: session-level current artifacts are distinct from run-record history.
|
|
||||||
|
|
||||||
## 2. Core Concepts
|
|
||||||
|
|
||||||
### 2.1 Session
|
|
||||||
|
|
||||||
A session is the stable unit of work identified by `campaign_id` and `session_id`.
|
|
||||||
|
|
||||||
Examples:
|
|
||||||
|
|
||||||
```text
|
|
||||||
campaign_id = dilfs
|
|
||||||
session_id = 2026-05-07
|
|
||||||
```
|
|
||||||
|
|
||||||
The session directory represents the current durable local state for that session. Re-running Narratio for the same session should consult this state, skip already-completed stages by default, and produce no changes unless work is incomplete, stale, forced, or explicitly selected.
|
|
||||||
|
|
||||||
### 2.2 Run
|
|
||||||
|
|
||||||
A run is a particular execution attempt identified by a generated `run_id`, for example:
|
|
||||||
|
|
||||||
```text
|
|
||||||
20260517T174748Z-abcd1234
|
|
||||||
```
|
|
||||||
|
|
||||||
A run may execute all stages or only a sparse subset of stages. Sparse runs are expected and desirable when the user invokes `--force`, `run-stage`, or a stage-limited command.
|
|
||||||
|
|
||||||
Run directories are provenance/debug records. They should reflect what actually happened during that invocation, not a synthetic complete pipeline layout.
|
|
||||||
|
|
||||||
### 2.3 Durable Output
|
|
||||||
|
|
||||||
A durable output is a canonical session-level artifact intended for later stages, user consumption, archive promotion, or future idempotency decisions.
|
|
||||||
|
|
||||||
Examples:
|
|
||||||
|
|
||||||
```text
|
|
||||||
transcripts/merged.json
|
|
||||||
transcripts/processed.json
|
|
||||||
transcripts/normalized.json
|
|
||||||
transcripts/trimmed.json
|
|
||||||
artifacts/session_recap.md
|
|
||||||
```
|
|
||||||
|
|
||||||
Durable outputs live directly under the session directory, not under a particular run directory.
|
|
||||||
|
|
||||||
### 2.4 Run-Local Output
|
|
||||||
|
|
||||||
A run-local output is the file initially produced by a stage during a specific run. After validation, durable outputs are promoted from run-local paths to session-level canonical paths.
|
|
||||||
|
|
||||||
Run-local outputs, logs, generated configs, reports, and scratch files should remain under:
|
|
||||||
|
|
||||||
```text
|
|
||||||
runs/{run_id}/{stage}/...
|
|
||||||
```
|
|
||||||
|
|
||||||
## 3. Local Workspace Layout
|
|
||||||
|
|
||||||
The canonical local workspace layout is:
|
|
||||||
|
|
||||||
```text
|
|
||||||
{workspace.root}/work/{campaign_id}/{session_id}/
|
|
||||||
manifest.json
|
|
||||||
current/
|
|
||||||
manifest.json
|
|
||||||
run_id.txt
|
|
||||||
inputs/
|
|
||||||
transcripts/
|
|
||||||
artifacts/
|
|
||||||
reports/
|
|
||||||
logs/
|
|
||||||
config/
|
|
||||||
runs/
|
|
||||||
{run_id}/
|
|
||||||
manifest.json
|
|
||||||
prepare/
|
|
||||||
transcribe/
|
|
||||||
merge/
|
|
||||||
polish/
|
|
||||||
normalize/
|
|
||||||
trim/
|
|
||||||
analyze/
|
|
||||||
archive/
|
|
||||||
notify/
|
|
||||||
```
|
|
||||||
|
|
||||||
Not every directory must exist at all times. Directories should be created idempotently when needed.
|
|
||||||
|
|
||||||
### 3.1 Session Root
|
|
||||||
|
|
||||||
The session root is:
|
|
||||||
|
|
||||||
```text
|
|
||||||
{workspace.root}/work/{campaign_id}/{session_id}/
|
|
||||||
```
|
|
||||||
|
|
||||||
The session root is the stable local home for the session. It is the default base for resolving canonical artifact paths.
|
|
||||||
|
|
||||||
The only files that should live directly in the session root are core session-state files, primarily:
|
|
||||||
|
|
||||||
```text
|
|
||||||
manifest.json
|
|
||||||
```
|
|
||||||
|
|
||||||
Lock files may also be session-root scoped if the implementation uses file locks there, but transient locks should not be treated as durable artifacts.
|
|
||||||
|
|
||||||
### 3.2 Session-Level Canonical Directories
|
|
||||||
|
|
||||||
The following directories contain current durable session state:
|
|
||||||
|
|
||||||
```text
|
|
||||||
inputs/
|
|
||||||
transcripts/
|
|
||||||
artifacts/
|
|
||||||
reports/
|
|
||||||
logs/
|
|
||||||
config/
|
|
||||||
current/
|
|
||||||
```
|
|
||||||
|
|
||||||
Recommended meanings:
|
|
||||||
|
|
||||||
| Directory | Purpose |
|
|
||||||
| -------------- | ----------------------------------------------------------------------------- |
|
|
||||||
| `inputs/` | Materialized or copied input files used by the current durable session state. |
|
|
||||||
| `transcripts/` | Canonical transcript tiers. |
|
|
||||||
| `artifacts/` | User-facing and machine-readable generated artifacts. |
|
|
||||||
| `reports/` | Canonical stage reports worth preserving at the session level. |
|
|
||||||
| `logs/` | Optional session-level logs or promoted/latest logs. |
|
|
||||||
| `config/` | Optional session-level generated config snapshots or promoted/latest configs. |
|
|
||||||
| `current/` | Current published session pointers, mirroring the archive backend. |
|
|
||||||
|
|
||||||
Canonical durable outputs should use stable paths under these directories.
|
|
||||||
|
|
||||||
### 3.3 Run History Directory
|
|
||||||
|
|
||||||
Run history lives under:
|
|
||||||
|
|
||||||
```text
|
|
||||||
{workspace.root}/work/{campaign_id}/{session_id}/runs/{run_id}/
|
|
||||||
```
|
|
||||||
|
|
||||||
Each run directory records what happened during that invocation. A run may contain all stage directories or only a sparse subset.
|
|
||||||
|
|
||||||
Example full run:
|
|
||||||
|
|
||||||
```text
|
|
||||||
runs/20260517T174748Z-abcd1234/
|
|
||||||
manifest.json
|
|
||||||
prepare/
|
|
||||||
transcribe/
|
|
||||||
merge/
|
|
||||||
polish/
|
|
||||||
normalize/
|
|
||||||
trim/
|
|
||||||
analyze/
|
|
||||||
archive/
|
|
||||||
notify/
|
|
||||||
```
|
|
||||||
|
|
||||||
Example sparse forced analyze run:
|
|
||||||
|
|
||||||
```text
|
|
||||||
runs/20260518T030000Z-efgh5678/
|
|
||||||
manifest.json
|
|
||||||
analyze/
|
|
||||||
```
|
|
||||||
|
|
||||||
Example sparse polish-through-analyze rerun:
|
|
||||||
|
|
||||||
```text
|
|
||||||
runs/20260518T041500Z-a1b2c3d4/
|
|
||||||
manifest.json
|
|
||||||
polish/
|
|
||||||
normalize/
|
|
||||||
trim/
|
|
||||||
analyze/
|
|
||||||
```
|
|
||||||
|
|
||||||
Run directories should not create stage folders for stages that were not selected, executed, skipped, or otherwise considered during that run unless there is a clear diagnostic reason to do so.
|
|
||||||
|
|
||||||
### 3.4 Stage Run-Local Directories
|
|
||||||
|
|
||||||
Each stage receives a run-local directory:
|
|
||||||
|
|
||||||
```text
|
|
||||||
runs/{run_id}/{stage}/
|
|
||||||
```
|
|
||||||
|
|
||||||
Within that stage directory, the stage may use subdirectories such as:
|
|
||||||
|
|
||||||
```text
|
|
||||||
outputs/
|
|
||||||
logs/
|
|
||||||
reports/
|
|
||||||
config/
|
|
||||||
scratch/
|
|
||||||
```
|
|
||||||
|
|
||||||
For example:
|
|
||||||
|
|
||||||
```text
|
|
||||||
runs/{run_id}/polish/
|
|
||||||
outputs/transcripts/processed.json
|
|
||||||
reports/audita.polish.report.json
|
|
||||||
logs/stdout.log
|
|
||||||
logs/stderr.log
|
|
||||||
config/audita.polish.generated.yml
|
|
||||||
scratch/
|
|
||||||
```
|
|
||||||
|
|
||||||
The exact internal layout of a stage directory may vary by stage, but it should be deterministic, documented, and generated through centralized path helpers rather than ad hoc path joins.
|
|
||||||
|
|
||||||
## 4. Promotion Model
|
|
||||||
|
|
||||||
Narratio uses stage-level promotion with immediate promotion after successful validation.
|
|
||||||
|
|
||||||
The stage lifecycle is:
|
|
||||||
|
|
||||||
1. Resolve required inputs from the current session state and/or run-local context.
|
|
||||||
2. Create the run-local stage directory.
|
|
||||||
3. Execute the stage, writing outputs under `runs/{run_id}/{stage}/...`.
|
|
||||||
4. Validate run-local outputs.
|
|
||||||
5. Promote durable outputs into session-level canonical paths.
|
|
||||||
6. Update the session manifest.
|
|
||||||
7. Update the run manifest.
|
|
||||||
|
|
||||||
Promotion means an atomic or effectively atomic copy/rename from a run-local path to a session-level canonical path.
|
|
||||||
|
|
||||||
Example:
|
|
||||||
|
|
||||||
```text
|
|
||||||
runs/{run_id}/polish/outputs/transcripts/processed.json
|
|
||||||
```
|
|
||||||
|
|
||||||
is promoted to:
|
|
||||||
|
|
||||||
```text
|
|
||||||
transcripts/processed.json
|
|
||||||
```
|
|
||||||
|
|
||||||
Promotion should be safe and deterministic:
|
|
||||||
|
|
||||||
* Validate before promotion.
|
|
||||||
* Write promoted files atomically where possible.
|
|
||||||
* Never leave partially written durable outputs.
|
|
||||||
* Record the producing `run_id` in the session manifest.
|
|
||||||
* Preserve run-local files for debugging unless retention policy deletes them.
|
|
||||||
|
|
||||||
## 5. Promotion Policy: Option A
|
|
||||||
|
|
||||||
Narratio uses immediate stage-level promotion.
|
|
||||||
|
|
||||||
If a selected stage succeeds, its durable outputs are promoted immediately, even if a later selected stage fails.
|
|
||||||
|
|
||||||
Example:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
narratio run --session-id 2026-05-07 --force --stages polish,normalize,trim,analyze
|
|
||||||
```
|
|
||||||
|
|
||||||
If `polish` succeeds and `normalize` fails:
|
|
||||||
|
|
||||||
* `transcripts/processed.json` may be updated from the new run.
|
|
||||||
* `normalize`, `trim`, and `analyze` should not be marked succeeded for the new input state.
|
|
||||||
* Downstream outputs may now be stale relative to the newly promoted polished transcript.
|
|
||||||
|
|
||||||
This policy is simpler, transparent, and consistent with stage-level resumability. It does require explicit stale/invalidation handling.
|
|
||||||
|
|
||||||
## 6. Stale and Invalidation Semantics
|
|
||||||
|
|
||||||
Full checksum-based stale detection may be implemented later. Before that exists, Narratio should still use a simple deterministic invalidation rule for forced or explicit upstream reruns.
|
|
||||||
|
|
||||||
When a stage is successfully re-executed and promoted, downstream stages should be marked stale unless they are also re-executed successfully in the same command invocation.
|
|
||||||
|
|
||||||
Example stage order:
|
|
||||||
|
|
||||||
```text
|
|
||||||
prepare -> transcribe -> merge -> polish -> normalize -> trim -> analyze -> archive -> notify
|
|
||||||
```
|
|
||||||
|
|
||||||
If `polish` is forced and promoted, then the following downstream stages should be invalidated unless rerun successfully:
|
|
||||||
|
|
||||||
```text
|
|
||||||
normalize
|
|
||||||
trim
|
|
||||||
analyze
|
|
||||||
archive
|
|
||||||
notify
|
|
||||||
```
|
|
||||||
|
|
||||||
A stale stage is not equivalent to a failed stage. It means its current durable outputs may no longer correspond to current upstream inputs or configuration.
|
|
||||||
|
|
||||||
Minimum manifest state model:
|
|
||||||
|
|
||||||
```text
|
|
||||||
pending
|
|
||||||
running
|
|
||||||
succeeded
|
|
||||||
failed
|
|
||||||
skipped
|
|
||||||
stale
|
|
||||||
```
|
|
||||||
|
|
||||||
If adding a new `stale` state is too invasive for v1.0, the implementation should at least record stale metadata or clear downstream success markers in a way that prevents accidental idempotent skips based on obsolete outputs.
|
|
||||||
|
|
||||||
## 7. Manifest Responsibilities
|
|
||||||
|
|
||||||
Narratio should distinguish between session manifests and run manifests.
|
|
||||||
|
|
||||||
The same underlying Go types may be reused where practical, but the concepts should remain separate.
|
|
||||||
|
|
||||||
### 7.1 Session Manifest
|
|
||||||
|
|
||||||
Path:
|
|
||||||
|
|
||||||
```text
|
|
||||||
{workspace.root}/work/{campaign_id}/{session_id}/manifest.json
|
|
||||||
```
|
|
||||||
|
|
||||||
The session manifest answers:
|
|
||||||
|
|
||||||
```text
|
|
||||||
What is the current durable state of this session?
|
|
||||||
```
|
|
||||||
|
|
||||||
It should record:
|
|
||||||
|
|
||||||
* campaign ID
|
|
||||||
* session ID
|
|
||||||
* current or latest run ID
|
|
||||||
* current stage states
|
|
||||||
* canonical durable output refs
|
|
||||||
* artifact IDs and paths
|
|
||||||
* producing run ID for each current stage output
|
|
||||||
* relevant input/config checksums when available
|
|
||||||
* stale/invalidated stage information
|
|
||||||
* archive/current publication metadata
|
|
||||||
|
|
||||||
A session's durable state may be a composite of multiple runs.
|
|
||||||
|
|
||||||
For example:
|
|
||||||
|
|
||||||
```text
|
|
||||||
transcripts/merged.json produced by run A
|
|
||||||
transcripts/processed.json produced by run B
|
|
||||||
transcripts/normalized.json produced by run B
|
|
||||||
transcripts/trimmed.json produced by run B
|
|
||||||
artifacts/session_recap.md produced by run C
|
|
||||||
```
|
|
||||||
|
|
||||||
This is valid and expected.
|
|
||||||
|
|
||||||
### 7.2 Run Manifest
|
|
||||||
|
|
||||||
Path:
|
|
||||||
|
|
||||||
```text
|
|
||||||
{workspace.root}/work/{campaign_id}/{session_id}/runs/{run_id}/manifest.json
|
|
||||||
```
|
|
||||||
|
|
||||||
The run manifest answers:
|
|
||||||
|
|
||||||
```text
|
|
||||||
What happened during this specific execution attempt?
|
|
||||||
```
|
|
||||||
|
|
||||||
It should record:
|
|
||||||
|
|
||||||
* run ID
|
|
||||||
* campaign ID
|
|
||||||
* session ID
|
|
||||||
* command mode and selected stages
|
|
||||||
* force flags or stage selection flags
|
|
||||||
* stages considered during this run
|
|
||||||
* stages executed during this run
|
|
||||||
* stages skipped during this run and reasons
|
|
||||||
* run-local output paths
|
|
||||||
* promoted output paths
|
|
||||||
* logs
|
|
||||||
* reports
|
|
||||||
* generated configs
|
|
||||||
* timings
|
|
||||||
* errors
|
|
||||||
* non-secret subprocess invocation metadata
|
|
||||||
|
|
||||||
Run manifests are primarily for debugging, auditability, and archive history.
|
|
||||||
|
|
||||||
## 8. Idempotency and Resume Behavior
|
|
||||||
|
|
||||||
The idempotency boundary is the session, not the run.
|
|
||||||
|
|
||||||
By default:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
narratio run --session-id 2026-05-07
|
|
||||||
```
|
|
||||||
|
|
||||||
should consult the session manifest and skip stages that are already succeeded and not stale.
|
|
||||||
|
|
||||||
If all stages are already complete, the command should execute zero stages and report that the session is already complete.
|
|
||||||
|
|
||||||
Forced execution creates a new run record but updates session-level durable state only for stages that actually succeed and promote outputs.
|
|
||||||
|
|
||||||
Examples:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
narratio run --session-id 2026-05-07 --force
|
|
||||||
```
|
|
||||||
|
|
||||||
Creates a new run and attempts to re-execute the selected/default stage set.
|
|
||||||
|
|
||||||
```bash
|
|
||||||
narratio run-stage --session-id 2026-05-07 analyze --force
|
|
||||||
```
|
|
||||||
|
|
||||||
Creates a sparse run that executes only `analyze`, then promotes updated analysis artifacts if successful.
|
|
||||||
|
|
||||||
```bash
|
|
||||||
narratio resume --session-id 2026-05-07
|
|
||||||
```
|
|
||||||
|
|
||||||
Uses the session manifest to determine what remains incomplete or stale. Resume does not need to resume the same `run_id` unless the implementation explicitly supports resuming an interrupted active run.
|
|
||||||
|
|
||||||
## 9. Artifact Resolution Contract
|
|
||||||
|
|
||||||
Narratio should provide a first-class artifact registry and resolver.
|
|
||||||
|
|
||||||
The resolver maps symbolic artifact source names to canonical session-level paths and manifest output kinds.
|
|
||||||
|
|
||||||
Stages and adapters should not hardcode path fragments when resolving cross-stage inputs. They should ask the artifact resolver for the current durable artifact by ID.
|
|
||||||
|
|
||||||
### 9.1 Canonical Artifact IDs
|
|
||||||
|
|
||||||
Preferred artifact IDs should be namespaced:
|
|
||||||
|
|
||||||
```text
|
|
||||||
narratio.transcript.merged
|
|
||||||
narratio.transcript.polished
|
|
||||||
narratio.transcript.full
|
|
||||||
narratio.transcript.trimmed
|
|
||||||
narratio.bounds.session
|
|
||||||
narratio.artifact.session_recap
|
|
||||||
```
|
|
||||||
|
|
||||||
Recommended initial registry:
|
|
||||||
|
|
||||||
| Artifact ID | Canonical Path | Producer Stage | Output Kind | Meaning |
|
|
||||||
| --------------------------------- | ------------------------------- | -------------- | ------------------------ | ------------------------------------- |
|
|
||||||
| `narratio.transcript.merged` | `transcripts/merged.json` | `merge` | `transcript_merged` | Deterministic Seriatim merge. |
|
|
||||||
| `narratio.transcript.polished` | `transcripts/processed.json` | `polish` | `transcript_processed` | Full Audita-polished transcript. |
|
|
||||||
| `narratio.transcript.full` | `transcripts/normalized.json` | `normalize` | `transcript_normalized` | Preferred full normalized transcript. |
|
|
||||||
| `narratio.transcript.trimmed` | `transcripts/trimmed.json` | `trim` | `transcript_trimmed` | Gameplay-only transcript. |
|
|
||||||
| `narratio.bounds.session` | `artifacts/session_bounds.json` | `trim` | `session_bounds` | Trim bounds selected for the session. |
|
|
||||||
| `narratio.artifact.session_recap` | `artifacts/session_recap.md` | `analyze` | `artifact_session_recap` | Generated session recap. |
|
|
||||||
|
|
||||||
### 9.2 Backward-Compatible Aliases
|
|
||||||
|
|
||||||
Existing source names should remain supported:
|
|
||||||
|
|
||||||
| Legacy Source | Preferred Artifact ID |
|
|
||||||
| ----------------------- | ------------------------------ |
|
|
||||||
| `processed_transcript` | `narratio.transcript.polished` |
|
|
||||||
| `normalized_transcript` | `narratio.transcript.full` |
|
|
||||||
| `trimmed_transcript` | `narratio.transcript.trimmed` |
|
|
||||||
|
|
||||||
These aliases may be supported silently for v1.0. Documentation should prefer namespaced IDs.
|
|
||||||
|
|
||||||
### 9.3 Resolver Behavior
|
|
||||||
|
|
||||||
Artifact resolution should follow this order:
|
|
||||||
|
|
||||||
1. Normalize aliases to canonical artifact IDs.
|
|
||||||
2. Look for a current output reference in the session manifest.
|
|
||||||
3. Fall back to the canonical session-level path.
|
|
||||||
4. If the artifact is required, fail clearly if missing.
|
|
||||||
5. If the artifact is optional and missing, omit it from the downstream invocation.
|
|
||||||
6. Validate the artifact using the expected content validator.
|
|
||||||
7. Return a resolved artifact record containing ID, path, producer stage, output kind, and provenance.
|
|
||||||
|
|
||||||
Example conceptual result:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"id": "narratio.transcript.trimmed",
|
|
||||||
"path": "/var/lib/narratio/work/dilfs/2026-05-07/transcripts/trimmed.json",
|
|
||||||
"producer_stage": "trim",
|
|
||||||
"producer_run_id": "20260517T174748Z-abcd1234",
|
|
||||||
"output_kind": "transcript_trimmed",
|
|
||||||
"content_type": "application/json"
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
### 9.4 Artifact Validation
|
|
||||||
|
|
||||||
Transcript artifacts must be valid JSON with a top-level `segments` array.
|
|
||||||
|
|
||||||
Markdown/text artifacts must exist and be non-empty when required.
|
|
||||||
|
|
||||||
Bounds artifacts must match the expected bounds schema and refer to segment IDs in the same transcript ID space used by the trim stage.
|
|
||||||
|
|
||||||
Validation should happen before a resolved artifact is passed to another stage or external subprocess.
|
|
||||||
|
|
||||||
## 10. Analyze Stage Implications
|
|
||||||
|
|
||||||
The analyze stage should consume artifacts through the artifact resolver.
|
|
||||||
|
|
||||||
Preferred Scriptorium config shape:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
scriptorium:
|
|
||||||
artifacts:
|
|
||||||
session_recap:
|
|
||||||
enabled: true
|
|
||||||
prompt_id: "dnd.session_recap"
|
|
||||||
output_path: "artifacts/session_recap.md"
|
|
||||||
inputs:
|
|
||||||
transcript:
|
|
||||||
source: "narratio.transcript.trimmed"
|
|
||||||
required: true
|
|
||||||
```
|
|
||||||
|
|
||||||
Additional artifacts can choose different transcript tiers:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
scriptorium:
|
|
||||||
artifacts:
|
|
||||||
table_summary:
|
|
||||||
enabled: true
|
|
||||||
prompt_id: "dnd.table_summary"
|
|
||||||
output_path: "artifacts/table_summary.md"
|
|
||||||
inputs:
|
|
||||||
transcript:
|
|
||||||
source: "narratio.transcript.full"
|
|
||||||
required: true
|
|
||||||
```
|
|
||||||
|
|
||||||
For v1.0, Narratio does not need a generic DAG engine. It may execute configured analyze artifacts in deterministic order and allow later artifacts to consume earlier artifacts only when that relationship is explicit and unambiguous.
|
|
||||||
|
|
||||||
Rules:
|
|
||||||
|
|
||||||
* Artifact inputs resolve from current session-level durable state.
|
|
||||||
* Outputs are first written run-locally.
|
|
||||||
* Successful analyze outputs are promoted to session-level `artifacts/` paths.
|
|
||||||
* Manifest output refs record the producing run ID.
|
|
||||||
* Optional inputs are omitted when unavailable.
|
|
||||||
* Required missing inputs fail before invoking Scriptorium.
|
|
||||||
|
|
||||||
## 11. Archive Alignment
|
|
||||||
|
|
||||||
Local workspace semantics should mirror archive semantics.
|
|
||||||
|
|
||||||
Local session-level durable paths:
|
|
||||||
|
|
||||||
```text
|
|
||||||
work/{campaign}/{session}/transcripts/trimmed.json
|
|
||||||
work/{campaign}/{session}/artifacts/session_recap.md
|
|
||||||
work/{campaign}/{session}/current/manifest.json
|
|
||||||
work/{campaign}/{session}/current/run_id.txt
|
|
||||||
work/{campaign}/{session}/runs/{run_id}/...
|
|
||||||
```
|
|
||||||
|
|
||||||
should map naturally to remote archive paths:
|
|
||||||
|
|
||||||
```text
|
|
||||||
{root_prefix}/campaigns/{campaign}/sessions/{session}/transcripts/trimmed.json
|
|
||||||
{root_prefix}/campaigns/{campaign}/sessions/{session}/artifacts/session_recap.md
|
|
||||||
{root_prefix}/campaigns/{campaign}/sessions/{session}/current/manifest.json
|
|
||||||
{root_prefix}/campaigns/{campaign}/sessions/{session}/current/run_id.txt
|
|
||||||
{root_prefix}/campaigns/{campaign}/sessions/{session}/runs/{run_id}/...
|
|
||||||
```
|
|
||||||
|
|
||||||
The archive stage should publish run records and promoted current artifacts consistently with the local model.
|
|
||||||
|
|
||||||
`current/run_id.txt` remains the effective commit marker for the archived current session state.
|
|
||||||
|
|
||||||
## 12. Path Helper Requirements
|
|
||||||
|
|
||||||
All code should use centralized path helpers for workspace paths.
|
|
||||||
|
|
||||||
Stage code should not manually assemble durable cross-stage paths using raw string joins except through the path model.
|
|
||||||
|
|
||||||
Recommended helper surface:
|
|
||||||
|
|
||||||
```text
|
|
||||||
SessionRoot(campaignID, sessionID)
|
|
||||||
SessionManifestPath(campaignID, sessionID)
|
|
||||||
SessionCurrentDir(campaignID, sessionID)
|
|
||||||
SessionTranscriptsDir(campaignID, sessionID)
|
|
||||||
SessionArtifactsDir(campaignID, sessionID)
|
|
||||||
SessionReportsDir(campaignID, sessionID)
|
|
||||||
SessionLogsDir(campaignID, sessionID)
|
|
||||||
SessionConfigDir(campaignID, sessionID)
|
|
||||||
RunsDir(campaignID, sessionID)
|
|
||||||
RunRoot(campaignID, sessionID, runID)
|
|
||||||
RunManifestPath(campaignID, sessionID, runID)
|
|
||||||
RunStageDir(campaignID, sessionID, runID, stage)
|
|
||||||
RunStageOutputsDir(campaignID, sessionID, runID, stage)
|
|
||||||
RunStageLogsDir(campaignID, sessionID, runID, stage)
|
|
||||||
RunStageReportsDir(campaignID, sessionID, runID, stage)
|
|
||||||
RunStageConfigDir(campaignID, sessionID, runID, stage)
|
|
||||||
CanonicalArtifactPath(campaignID, sessionID, artifactID)
|
|
||||||
```
|
|
||||||
|
|
||||||
Path helpers should enforce safe relative paths for configured output paths:
|
|
||||||
|
|
||||||
* reject absolute paths unless explicitly allowed for a particular config field
|
|
||||||
* reject `..` traversal
|
|
||||||
* normalize separators
|
|
||||||
* preserve deterministic output paths
|
|
||||||
|
|
||||||
## 13. Directory Creation Policy
|
|
||||||
|
|
||||||
Directory creation should be centralized and idempotent.
|
|
||||||
|
|
||||||
Recommended policy:
|
|
||||||
|
|
||||||
* `prepare` ensures the baseline session directory structure exists.
|
|
||||||
* Every stage also calls shared layout helpers to ensure its required run-local directories exist before writing.
|
|
||||||
* `run-stage` should not depend on a prior `prepare` invocation merely to create folders.
|
|
||||||
* Missing directories should be created with appropriate permissions.
|
|
||||||
* Directory creation should not imply stage success.
|
|
||||||
|
|
||||||
This provides consistent layout while keeping direct stage execution robust.
|
|
||||||
|
|
||||||
## 14. Cleanup and Retention
|
|
||||||
|
|
||||||
Cleanup must preserve the distinction between durable session state and run history.
|
|
||||||
|
|
||||||
Workspace cleanup after successful archive may remove selected local directories only according to explicit configuration.
|
|
||||||
|
|
||||||
Potential retention policies:
|
|
||||||
|
|
||||||
```text
|
|
||||||
keep_all_runs
|
|
||||||
keep_failed_runs
|
|
||||||
keep_last_n_runs
|
|
||||||
delete_run_after_success
|
|
||||||
```
|
|
||||||
|
|
||||||
For v1.0, conservative retention is preferred:
|
|
||||||
|
|
||||||
* Do not delete durable session-level outputs unless explicitly requested.
|
|
||||||
* Do not delete failed run directories by default.
|
|
||||||
* If cleanup is enabled, remove only documented run-scoped or spool-scoped paths.
|
|
||||||
* Local development audio inputs must never be deleted by workspace cleanup.
|
|
||||||
|
|
||||||
## 15. Canonical-Only Layout Policy
|
|
||||||
|
|
||||||
Narratio now supports only the canonical campaign-aware layout:
|
|
||||||
|
|
||||||
```text
|
|
||||||
{workspace.root}/work/{campaign_id}/{session_id}/manifest.json
|
|
||||||
{workspace.root}/work/{campaign_id}/{session_id}/runs/{run_id}/...
|
|
||||||
```
|
|
||||||
|
|
||||||
Legacy session-only layout compatibility is intentionally not implemented.
|
|
||||||
|
|
||||||
If legacy workspace data exists, operators should recreate or manually migrate that data outside Narratio before running v1.0 commands.
|
|
||||||
|
|
||||||
## 16. Documentation Updates Required
|
|
||||||
|
|
||||||
The following documentation should be updated to reflect this architecture:
|
|
||||||
|
|
||||||
* `README.md`
|
|
||||||
* `docs/architecture.md`
|
|
||||||
* a dedicated workspace/run-history document, such as this file
|
|
||||||
* S3/archive documentation
|
|
||||||
* analyze/artifact configuration documentation
|
|
||||||
* example pipeline files
|
|
||||||
|
|
||||||
Documentation should consistently use the following terms:
|
|
||||||
|
|
||||||
| Term | Meaning |
|
|
||||||
| ---------------- | ---------------------------------------------------------------------- |
|
|
||||||
| Session | Durable domain object and idempotency boundary. |
|
|
||||||
| Run | Execution attempt that may update session state. |
|
|
||||||
| Durable output | Canonical current session-level output. |
|
|
||||||
| Run-local output | Output produced inside a specific run directory before promotion. |
|
|
||||||
| Promotion | Validated copy/rename from run-local output to durable session output. |
|
|
||||||
| Session manifest | Current durable state of the session. |
|
|
||||||
| Run manifest | Execution record for a particular run. |
|
|
||||||
| Artifact ID | Symbolic source name resolved by the artifact registry. |
|
|
||||||
|
|
||||||
## 17. Architectural Invariants
|
|
||||||
|
|
||||||
The following invariants should hold after implementation:
|
|
||||||
|
|
||||||
1. `session_id` remains the idempotency boundary for normal operator commands.
|
|
||||||
2. `run_id` identifies an execution attempt, not the primary durable workspace.
|
|
||||||
3. Session-level canonical artifacts are the default inputs for downstream stages.
|
|
||||||
4. Run-local outputs are promoted only after validation.
|
|
||||||
5. A session's current durable state may be composed of outputs from multiple runs.
|
|
||||||
6. Sparse run directories are valid and expected.
|
|
||||||
7. The session manifest records current stage/artifact state and producer run IDs.
|
|
||||||
8. The run manifest records what happened during one invocation.
|
|
||||||
9. Artifact consumers resolve symbolic artifact IDs through a registry/resolver.
|
|
||||||
10. Local workspace semantics mirror S3 archive semantics.
|
|
||||||
11. Directory creation is centralized and idempotent.
|
|
||||||
12. Stage code uses path helpers rather than ad hoc path construction.
|
|
||||||
13. Forced upstream reruns invalidate downstream stage success unless downstream stages are rerun successfully.
|
|
||||||
14. Cleanup never removes durable session outputs or local development inputs unless explicitly configured to do so.
|
|
||||||
|
|
||||||
## 18. Implementation Guidance
|
|
||||||
|
|
||||||
A practical implementation sequence is:
|
|
||||||
|
|
||||||
1. Add this architecture document.
|
|
||||||
2. Add or revise path model helpers for session roots, run roots, stage directories, and canonical artifact paths.
|
|
||||||
3. Introduce session manifest versus run manifest concepts.
|
|
||||||
4. Route stage outputs through run-local directories.
|
|
||||||
5. Add promotion helpers with validation and atomic writes.
|
|
||||||
6. Update existing stages to promote durable outputs to session-level canonical paths.
|
|
||||||
7. Add artifact registry and resolver.
|
|
||||||
8. Update analyze to use artifact IDs and aliases.
|
|
||||||
9. Add simple downstream stale invalidation for forced upstream reruns.
|
|
||||||
10. Align archive/local path behavior and documentation.
|
|
||||||
11. Update examples and README.
|
|
||||||
12. Add tests for idempotency, sparse forced runs, promotion, manifest provenance, and artifact resolution.
|
|
||||||
|
|
||||||
This sequence intentionally avoids introducing a generic DAG engine. The v1.0 goal is a clear, deterministic, stage-oriented orchestrator with stable session-level outputs and inspectable run history.
|
|
||||||
356
docs/documentation/policy.md
Normal file
356
docs/documentation/policy.md
Normal file
@@ -0,0 +1,356 @@
|
|||||||
|
# Go Project Documentation Policy
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
|
||||||
|
Project documentation must help four audiences:
|
||||||
|
|
||||||
|
1. users who need to run the application;
|
||||||
|
2. administrators/operators who need to configure and operate it;
|
||||||
|
3. developers who need to understand and change it safely;
|
||||||
|
4. LLM coding agents that need clear scope, boundaries, and invariants.
|
||||||
|
|
||||||
|
Docs should be accurate, concise, task-oriented, and organized by audience. Prefer links to canonical docs over repetition.
|
||||||
|
|
||||||
|
## Core Rules
|
||||||
|
|
||||||
|
### 1. Keep docs concise
|
||||||
|
|
||||||
|
Each document should cover a defined scope and only the essentials for that scope.
|
||||||
|
|
||||||
|
Avoid:
|
||||||
|
- long background explanations;
|
||||||
|
- repeated reference material;
|
||||||
|
- implementation detail in user-facing docs;
|
||||||
|
- aspirational language outside roadmap docs;
|
||||||
|
- verbose examples where one minimal example is clearer.
|
||||||
|
|
||||||
|
### 2. Document only implemented behavior outside roadmap files
|
||||||
|
|
||||||
|
Unimplemented, planned, aspirational, experimental, or future work may be described only under:
|
||||||
|
|
||||||
|
- `docs/roadmap/`
|
||||||
|
|
||||||
|
No other documentation file, including `README.md`, should describe code, features, modules, stages, commands, config fields, or behaviors that do not currently exist.
|
||||||
|
|
||||||
|
If a feature is partial, non-roadmap docs may describe only the implemented portion and its current boundary.
|
||||||
|
|
||||||
|
### 3. Use canonical homes
|
||||||
|
|
||||||
|
Each type of information should have one canonical location.
|
||||||
|
|
||||||
|
Canonical homes:
|
||||||
|
|
||||||
|
- project purpose and quickstart: `README.md`
|
||||||
|
- development principles: `docs/architecture.md`
|
||||||
|
- configuration reference: `docs/config.md`
|
||||||
|
- CLI reference: `docs/cli.md`
|
||||||
|
- operations and recovery: `docs/operations.md`
|
||||||
|
- troubleshooting: `docs/troubleshooting.md`
|
||||||
|
- implemented internals: `docs/internal/`
|
||||||
|
- future work: `docs/roadmap/`
|
||||||
|
- contributor workflow: `docs/development.md`
|
||||||
|
- copyable examples: `examples/`
|
||||||
|
|
||||||
|
Other files should summarize briefly and link to the canonical source.
|
||||||
|
|
||||||
|
### 4. Keep examples real
|
||||||
|
|
||||||
|
Examples should be valid, maintained, and free of secrets.
|
||||||
|
|
||||||
|
Where practical:
|
||||||
|
- example configs should load successfully;
|
||||||
|
- example commands should match real CLI syntax;
|
||||||
|
- important examples should be covered by tests.
|
||||||
|
|
||||||
|
## Documentation Profiles
|
||||||
|
|
||||||
|
All projects require:
|
||||||
|
|
||||||
|
- `README.md`
|
||||||
|
- `docs/architecture.md`
|
||||||
|
|
||||||
|
Additional docs depend on the project.
|
||||||
|
|
||||||
|
### Small library
|
||||||
|
|
||||||
|
Recommended:
|
||||||
|
- `docs/development.md`, if contributor conventions are non-obvious
|
||||||
|
|
||||||
|
### Simple CLI
|
||||||
|
|
||||||
|
Required:
|
||||||
|
- `docs/cli.md`
|
||||||
|
|
||||||
|
Recommended:
|
||||||
|
- `docs/development.md`
|
||||||
|
|
||||||
|
### Config-driven CLI
|
||||||
|
|
||||||
|
Required:
|
||||||
|
- `docs/cli.md`
|
||||||
|
- `docs/config.md`
|
||||||
|
|
||||||
|
Recommended:
|
||||||
|
- `examples/`
|
||||||
|
- `docs/development.md`
|
||||||
|
|
||||||
|
### Stateful or operator-facing application
|
||||||
|
|
||||||
|
Required:
|
||||||
|
- `docs/cli.md`, if CLI-based
|
||||||
|
- `docs/config.md`, if config-driven
|
||||||
|
- `docs/operations.md`
|
||||||
|
|
||||||
|
Recommended:
|
||||||
|
- `docs/troubleshooting.md`
|
||||||
|
- `examples/`
|
||||||
|
- `docs/development.md`
|
||||||
|
|
||||||
|
### Modular, staged, service-oriented, or orchestration application
|
||||||
|
|
||||||
|
Required:
|
||||||
|
- `docs/cli.md`, if CLI-based
|
||||||
|
- `docs/config.md`, if config-driven
|
||||||
|
- `docs/operations.md`
|
||||||
|
- `docs/internal/`
|
||||||
|
- `docs/development.md`
|
||||||
|
|
||||||
|
Recommended:
|
||||||
|
- `docs/troubleshooting.md`
|
||||||
|
- validated examples under `examples/`
|
||||||
|
|
||||||
|
## Required Documents
|
||||||
|
|
||||||
|
### README.md
|
||||||
|
|
||||||
|
**Audience:** users, administrators, operators
|
||||||
|
|
||||||
|
The README is the outward-facing project orientation page.
|
||||||
|
|
||||||
|
It should include, in order:
|
||||||
|
|
||||||
|
1. concise description;
|
||||||
|
2. elevator pitch;
|
||||||
|
3. shortest useful command or usage example;
|
||||||
|
4. links to targeted docs.
|
||||||
|
|
||||||
|
The README should be short. It is not a manual.
|
||||||
|
|
||||||
|
The “shortest useful command” means the simplest command that performs the project’s core use case. (It does not mean `app --help`.)
|
||||||
|
|
||||||
|
### docs/architecture.md
|
||||||
|
|
||||||
|
**Audience:** developers, LLM coding agents
|
||||||
|
|
||||||
|
`docs/architecture.md` is required for every project.
|
||||||
|
|
||||||
|
It is an inward-facing development policy document. It should describe how the project is intended to be built and changed.
|
||||||
|
|
||||||
|
It should include:
|
||||||
|
|
||||||
|
- project shape;
|
||||||
|
- core design principles;
|
||||||
|
- package and boundary philosophy;
|
||||||
|
- state/persistence philosophy, if applicable;
|
||||||
|
- external integration philosophy, if applicable;
|
||||||
|
- error-handling and logging principles;
|
||||||
|
- testing expectations;
|
||||||
|
- documentation expectations;
|
||||||
|
- architectural invariants;
|
||||||
|
- explicit non-goals, if useful.
|
||||||
|
|
||||||
|
For small projects, this file may be brief. It may simply state that the project is intentionally narrow, monolithic, and dependency-light.
|
||||||
|
|
||||||
|
### docs/config.md
|
||||||
|
|
||||||
|
**Audience:** administrators, operators, advanced users
|
||||||
|
|
||||||
|
Required for applications with configuration files.
|
||||||
|
|
||||||
|
It should include, in order:
|
||||||
|
|
||||||
|
1. config file locations and discovery precedence;
|
||||||
|
2. minimal working config;
|
||||||
|
3. production-oriented config;
|
||||||
|
4. full configuration reference;
|
||||||
|
5. secrets handling, if applicable;
|
||||||
|
6. links to maintained examples.
|
||||||
|
|
||||||
|
The full configuration reference should be canonical.
|
||||||
|
|
||||||
|
### docs/cli.md
|
||||||
|
|
||||||
|
**Audience:** users, administrators, operators
|
||||||
|
|
||||||
|
Required for CLI applications.
|
||||||
|
|
||||||
|
It should include, in order:
|
||||||
|
|
||||||
|
1. shortest useful command;
|
||||||
|
2. command overview;
|
||||||
|
3. complete flag reference;
|
||||||
|
4. common workflows;
|
||||||
|
5. diagnostic or recovery commands, if applicable.
|
||||||
|
|
||||||
|
Explain when commands are useful, not just their syntax.
|
||||||
|
|
||||||
|
### docs/operations.md
|
||||||
|
|
||||||
|
**Audience:** administrators, operators
|
||||||
|
|
||||||
|
Required for applications that maintain state, support resume behavior, run multiple stages, write durable artifacts, use remote storage, or require recovery procedures.
|
||||||
|
|
||||||
|
It should cover:
|
||||||
|
|
||||||
|
- normal workflow;
|
||||||
|
- filesystem layout;
|
||||||
|
- remote storage layout, if applicable;
|
||||||
|
- logs and manifests;
|
||||||
|
- resume/retry behavior;
|
||||||
|
- cleanup behavior;
|
||||||
|
- archive/backup behavior;
|
||||||
|
- safe recovery procedures;
|
||||||
|
- operational caveats.
|
||||||
|
|
||||||
|
### docs/troubleshooting.md
|
||||||
|
|
||||||
|
**Audience:** administrators, operators
|
||||||
|
|
||||||
|
Recommended once recurring failure modes exist.
|
||||||
|
|
||||||
|
Each entry should include:
|
||||||
|
|
||||||
|
- symptom;
|
||||||
|
- likely cause;
|
||||||
|
- diagnostic command or inspection step;
|
||||||
|
- safe fix;
|
||||||
|
- relevant links.
|
||||||
|
|
||||||
|
### docs/development.md
|
||||||
|
|
||||||
|
**Audience:** developers, LLM coding agents
|
||||||
|
|
||||||
|
Required for projects maintained by humans and LLM coding agents.
|
||||||
|
|
||||||
|
It should include:
|
||||||
|
|
||||||
|
- repository layout;
|
||||||
|
- build/test commands;
|
||||||
|
- coding conventions;
|
||||||
|
- dependency policy;
|
||||||
|
- how to add config fields;
|
||||||
|
- how to add CLI flags;
|
||||||
|
- how to add stages/modules/adapters, if applicable;
|
||||||
|
- how to update examples;
|
||||||
|
- documentation update expectations.
|
||||||
|
|
||||||
|
### docs/internal/
|
||||||
|
|
||||||
|
**Audience:** developers, LLM coding agents
|
||||||
|
|
||||||
|
Required for modular, staged, service-oriented, or orchestration projects.
|
||||||
|
|
||||||
|
This directory describes implemented internal components. It is not the roadmap.
|
||||||
|
|
||||||
|
Use one file per major component where useful.
|
||||||
|
|
||||||
|
Each component doc should include:
|
||||||
|
|
||||||
|
1. purpose;
|
||||||
|
2. inputs and outputs;
|
||||||
|
3. boundaries;
|
||||||
|
4. config fields used;
|
||||||
|
5. external adapters used;
|
||||||
|
6. state or manifest behavior, if applicable;
|
||||||
|
7. skip/resume behavior, if applicable;
|
||||||
|
8. failure behavior;
|
||||||
|
9. tests to inspect before changing;
|
||||||
|
10. architectural invariants.
|
||||||
|
|
||||||
|
### docs/roadmap/
|
||||||
|
|
||||||
|
**Audience:** maintainers, developers, LLM coding agents
|
||||||
|
|
||||||
|
This is the only place for planned, future, aspirational, experimental, or unimplemented work.
|
||||||
|
|
||||||
|
Roadmap docs should clearly distinguish:
|
||||||
|
|
||||||
|
- proposed work;
|
||||||
|
- accepted plans;
|
||||||
|
- deferred ideas;
|
||||||
|
- rejected ideas;
|
||||||
|
- implementation prompts or task breakdowns, if useful.
|
||||||
|
|
||||||
|
Roadmap docs should not be confused with current behavior.
|
||||||
|
|
||||||
|
### docs/integrations/
|
||||||
|
|
||||||
|
**Audience:** developers, LLM coding agents
|
||||||
|
|
||||||
|
Required for projects that depend on external CLIs, APIs, services, protocols, or file formats where the integration contract is important to maintain.
|
||||||
|
|
||||||
|
This directory contains concise, versioned reference notes for external integration contracts. It should document only the parts of the external system that this project actually uses.
|
||||||
|
|
||||||
|
Use one file per integration where useful.
|
||||||
|
|
||||||
|
## Examples Directory
|
||||||
|
|
||||||
|
Projects with non-trivial configuration or workflows should include `examples/`.
|
||||||
|
|
||||||
|
Useful examples include:
|
||||||
|
|
||||||
|
- minimal working config;
|
||||||
|
- production-oriented config;
|
||||||
|
- full annotated config;
|
||||||
|
- local development config;
|
||||||
|
- remote/object-storage config;
|
||||||
|
- minimal session/input file.
|
||||||
|
|
||||||
|
Examples should be valid, maintained, tested when practical, and linked from relevant docs.
|
||||||
|
|
||||||
|
## Security and Privacy
|
||||||
|
|
||||||
|
Docs and examples must not include:
|
||||||
|
|
||||||
|
- real API keys;
|
||||||
|
- tokens;
|
||||||
|
- passwords;
|
||||||
|
- private keys;
|
||||||
|
- private environment dumps;
|
||||||
|
- sensitive user data;
|
||||||
|
- raw private transcripts;
|
||||||
|
- private infrastructure details unless intentionally public.
|
||||||
|
|
||||||
|
Document secret-handling mechanisms, not actual secret values.
|
||||||
|
|
||||||
|
## Maintenance Rules
|
||||||
|
|
||||||
|
When docs change, verify the affected behavior.
|
||||||
|
|
||||||
|
Where practical:
|
||||||
|
|
||||||
|
- load example config files in tests;
|
||||||
|
- test CLI examples or command parser behavior;
|
||||||
|
- validate documented flags against real flags;
|
||||||
|
- remove stale references;
|
||||||
|
- update links after renames;
|
||||||
|
- keep roadmap content out of non-roadmap docs.
|
||||||
|
|
||||||
|
If documentation and code disagree, fix the documentation and/or open a roadmap item; do not leave aspirational behavior in current-behavior docs.
|
||||||
|
|
||||||
|
Documentation is complete only when it matches the current code.
|
||||||
|
|
||||||
|
## Documentation Change Checklist
|
||||||
|
|
||||||
|
Before merging documentation changes, verify:
|
||||||
|
|
||||||
|
- README is concise and orientation-focused.
|
||||||
|
- `docs/architecture.md` describes development principles.
|
||||||
|
- Future work appears only under `docs/roadmap/`.
|
||||||
|
- User-facing docs avoid unnecessary internals.
|
||||||
|
- Developer-facing docs preserve boundaries and invariants.
|
||||||
|
- Config examples match the schema.
|
||||||
|
- CLI examples match real commands and flags.
|
||||||
|
- Defaults appear in the canonical config reference.
|
||||||
|
- No secrets or private data are included.
|
||||||
|
- Links are accurate.
|
||||||
15
docs/integrations/README.md
Normal file
15
docs/integrations/README.md
Normal file
@@ -0,0 +1,15 @@
|
|||||||
|
# Integration Documentation Index
|
||||||
|
|
||||||
|
## Audience
|
||||||
|
Developers and LLM coding agents changing Narratio's external integration contracts.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
Implemented-only reference notes for the external systems Narratio currently integrates with.
|
||||||
|
|
||||||
|
## Integration Docs
|
||||||
|
- `audita.md`: Audita adapter invocation and validation contract.
|
||||||
|
- `seriatim.md`: Seriatim normalize/merge/trim adapter contract.
|
||||||
|
- `scriptorium.md`: Scriptorium run/render adapter contract.
|
||||||
|
|
||||||
|
## Canonical Owner
|
||||||
|
`docs/integrations/` is the canonical home for external integration reference notes per `docs/documentation/policy.md`.
|
||||||
@@ -1,96 +1,66 @@
|
|||||||
# Audita Subprocess Operations
|
# Integration: audita
|
||||||
|
|
||||||
This document describes how parent processes should invoke `audita process` safely in production orchestration.
|
## Purpose
|
||||||
|
Define Narratio's adapter contract for transcript polishing via Audita CLI subprocess execution.
|
||||||
|
|
||||||
## Recommended command form
|
## Inputs and Outputs
|
||||||
|
Inputs (`audita.PolishRequest`):
|
||||||
|
- merged transcript path
|
||||||
|
- glossary path
|
||||||
|
- output processed transcript path
|
||||||
|
- optional report path (required when report enabled)
|
||||||
|
- work dir
|
||||||
|
- generated config path
|
||||||
|
- stdout/stderr log paths
|
||||||
|
- optional module/model/base URL and concurrency knobs
|
||||||
|
|
||||||
Use explicit file outputs for orchestrated runs:
|
Outputs (`audita.PolishResult`):
|
||||||
|
- processed transcript path
|
||||||
|
- optional report path
|
||||||
|
- generated config path
|
||||||
|
- stdout/stderr log paths
|
||||||
|
- exit code, duration, invoked binary
|
||||||
|
- adapter metadata
|
||||||
|
|
||||||
```sh
|
## Boundaries
|
||||||
audita process <transcript.json> \
|
Owns:
|
||||||
--transcript-description "Brief context that may help resolve ambiguous terms." \
|
- Deterministic CLI argument construction for `audita process`
|
||||||
--glossary <glossary.yaml> \
|
- Environment bridging for API credentials
|
||||||
--output <output-transcript.json> \
|
- Invocation config emission
|
||||||
--report-json <report.json>
|
- Output validation for processed transcript and report
|
||||||
```
|
|
||||||
|
|
||||||
Additional flags that may be situationally appropriate:
|
Does not own:
|
||||||
- `--config <path>` to select an explicit versioned config file.
|
- Upstream/downstream stage orchestration
|
||||||
- `--output-schema <bare-segments|audita-v1>` to select transcript output shape.
|
- Credential sourcing policy beyond required env-var presence check
|
||||||
- `--work-dir <dir>` to control diagnostics location.
|
|
||||||
- `--work-dir-retention <always|auto|never>` to control retained run directories.
|
|
||||||
- `--total-llm-concurrency`, `--proposal-llm-concurrency`, and `--validation-llm-concurrency` when orchestration needs to set explicit LLM throughput controls.
|
|
||||||
- `--modules ...` only when intentionally overriding the default sequence.
|
|
||||||
|
|
||||||
For config-driven orchestration, validate config files in CI/preflight:
|
## Config Fields Used
|
||||||
|
Via `pipeline.audita.*` mapped in app/stage wiring:
|
||||||
|
- `binary`, `timeout`, `llm_api_key_env`, `modules`, `base_url`, `model`
|
||||||
|
- `transcript_description`, `config_path`, `output_schema`, `work_dir_retention`
|
||||||
|
- `total_llm_concurrency`, `proposal_llm_concurrency`, `validation_model`, `validation_llm_concurrency`, `report`
|
||||||
|
|
||||||
```sh
|
## External Adapters Used
|
||||||
audita config validate --config <path>
|
- Shared subprocess helper (`internal/adapters/subprocess`) to run CLI and capture logs.
|
||||||
```
|
|
||||||
|
|
||||||
## Stdout behavior
|
## State and Manifest Behavior
|
||||||
|
- No direct manifest writes.
|
||||||
|
- Stage-level metadata records adapter provenance and credential-present signal.
|
||||||
|
- Generated invocation YAML is written when `GeneratedConfigPath` is provided.
|
||||||
|
|
||||||
- With `--output`: stdout is expected to be empty on success.
|
## Skip and Resume Behavior
|
||||||
- Without `--output`: stdout contains transcript JSON only on success.
|
- Adapter has no skip/resume logic. Stage/runner controls this.
|
||||||
- Report JSON is never written to stdout.
|
|
||||||
|
|
||||||
## Stderr behavior
|
## Failure Behavior
|
||||||
|
- Constructor validation fails on invalid binary/timeout/schema/concurrency/URL values.
|
||||||
|
- Run fails on missing required paths, missing required credential env var, subprocess errors, invalid processed JSON shape, or invalid report JSON.
|
||||||
|
- Failures preserve stdout/stderr paths in returned result metadata.
|
||||||
|
|
||||||
- Success path should be quiet or minimal human-readable logs.
|
## Tests to Inspect Before Changing
|
||||||
- Failure path writes concise human-readable errors.
|
- `internal/adapters/audita/subprocess_test.go`
|
||||||
- When a diagnostics run directory exists, failure stderr includes its path.
|
- `internal/adapters/audita/fake_test.go`
|
||||||
- Prompt/response diagnostic payloads are not streamed to stderr.
|
- `internal/stage/polish_test.go`
|
||||||
|
|
||||||
## Output file behavior
|
## Architectural Invariants
|
||||||
|
- Processed output must be valid JSON with top-level `segments` array.
|
||||||
- `--output` writes transcript JSON in the selected output schema to the provided path.
|
- When report is enabled, report output must be valid JSON.
|
||||||
- Output write failures return nonzero and surface actionable errors.
|
- If `llm_api_key_env` is configured, credential must be present in environment.
|
||||||
- The command does not silently ignore output write errors.
|
|
||||||
|
|
||||||
## Report JSON behavior
|
|
||||||
|
|
||||||
- `--report-json` writes a machine-readable process report to the requested path.
|
|
||||||
- Run-directory `report.json` is written independently under diagnostics.
|
|
||||||
- Best-effort failure reports are emitted when possible without masking the primary failure.
|
|
||||||
- Report write failures return nonzero with clear stderr messaging.
|
|
||||||
- Report diagnostics metadata references run-directory artifacts including utilization diagnostics and correction ledger paths when available.
|
|
||||||
|
|
||||||
## Diagnostics directory behavior
|
|
||||||
|
|
||||||
- Each run creates (when possible) a per-run diagnostics directory.
|
|
||||||
- Typical artifacts include transcript, normalization, chunking, invocation, effective config, LLM diagnostics, `utilization-diagnostics.json`, `correction-ledger.json`, `report.json`, and `error.log` on failure.
|
|
||||||
- Failed runs retain diagnostics.
|
|
||||||
- Under `auto` retention, successful runs with skipped/rejected corrections are retained; clean successful runs may be removed.
|
|
||||||
|
|
||||||
## Exit codes
|
|
||||||
|
|
||||||
- `0`: success.
|
|
||||||
- Nonzero: failure (input/schema/config/module/LLM/runtime/output/report/diagnostics errors).
|
|
||||||
|
|
||||||
Treat any nonzero as a failed subprocess invocation.
|
|
||||||
|
|
||||||
## Timeout and cancellation
|
|
||||||
|
|
||||||
- Runtime operations propagate context cancellation and request timeouts through LLM/scheduler paths.
|
|
||||||
- On cancellation or timeout, the process exits nonzero and should not hang.
|
|
||||||
- If diagnostics were initialized before failure, failure artifacts remain available for debugging.
|
|
||||||
|
|
||||||
## Secret redaction expectations
|
|
||||||
|
|
||||||
API keys and configured secret values are redacted from:
|
|
||||||
- reports (`--report-json` and run-dir `report.json`);
|
|
||||||
- diagnostics artifacts (including effective config and LLM interaction artifacts);
|
|
||||||
- surfaced adapter/runtime errors;
|
|
||||||
- test fixtures and regression outputs.
|
|
||||||
|
|
||||||
Parent-process logs should still avoid printing raw environment variables.
|
|
||||||
|
|
||||||
## Parent-process pipe guidance
|
|
||||||
|
|
||||||
To avoid deadlocks in orchestrators:
|
|
||||||
- always read both stdout and stderr concurrently when invoking as a subprocess;
|
|
||||||
- prefer file outputs (`--output`, `--report-json`) for machine workflows;
|
|
||||||
- treat stderr as human-readable diagnostics, not structured data;
|
|
||||||
- parse structured results from output/report files.
|
|
||||||
|
|
||||||
For Go callers, prefer `exec.CommandContext` with explicit timeout/cancellation and buffered/streamed readers for both pipes.
|
|
||||||
|
|||||||
@@ -1,339 +1,64 @@
|
|||||||
# Narratio -> Scriptorium CLI Integration
|
# Integration: scriptorium
|
||||||
|
|
||||||
## 1. Purpose
|
## Purpose
|
||||||
|
Define Narratio's adapter contract for Scriptorium artifact generation and render-debug subprocess invocations.
|
||||||
This document defines how Narratio should invoke Scriptorium through the **public CLI**.
|
|
||||||
|
## Inputs and Outputs
|
||||||
This is a **subprocess integration contract**, not an internal Go API contract.
|
Inputs:
|
||||||
|
- `RunArtifactRequest`: binary, config path, prompt/profile IDs, input map, vars map, timeout, output path, logs/config paths, optional API env and working dir
|
||||||
## 2. Assumptions
|
- `RenderArtifactRequest`: same core fields for render mode
|
||||||
|
|
||||||
- `scriptorium` is installed and available on `PATH`.
|
Outputs (`ArtifactResult`):
|
||||||
- Scriptorium is configured with `config.yml`.
|
- output path
|
||||||
- `config.yml` provides `prompt_dir`, `profile_dir`, and `schema_dir` as needed.
|
- stdout/stderr log paths
|
||||||
- Prompt and profile libraries are already deployed for the environment.
|
- generated config path
|
||||||
- Narratio provides prepared artifact files (for example polished transcript, glossary, previous recap, campaign notes).
|
- exit code and duration
|
||||||
- Initial integration is synchronous subprocess execution.
|
- command mode (`run` or `render`)
|
||||||
- Narratio remains the orchestrator.
|
- prompt/profile provenance
|
||||||
|
- validation failure signal
|
||||||
In normal operation, Narratio does not need to pass `--prompt-dir` and `--profile-dir` if they are supplied by Scriptorium config.
|
- adapter metadata
|
||||||
|
|
||||||
Narratio may pass `--config <PATH>` when it must use a non-default Scriptorium config file.
|
## Boundaries
|
||||||
|
Owns:
|
||||||
## 3. Core Commands Narratio May Call
|
- Deterministic CLI arg construction for `scriptorium run` and `scriptorium render`
|
||||||
|
- Common request validation
|
||||||
Primary commands for subprocess integration:
|
- Invocation config emission
|
||||||
|
- Output existence/non-empty checks
|
||||||
- `scriptorium run`
|
- Validation-failure mapping for run exit code 2
|
||||||
- `scriptorium render`
|
|
||||||
|
Does not own:
|
||||||
For production generation, use `scriptorium run`.
|
- Artifact selection policy (`analyze` stage)
|
||||||
|
- Bounds semantic validation (`trim` stage)
|
||||||
`scriptorium render` is for debugging, dry-runs, test assertions, and validating command construction without LLM execution.
|
|
||||||
|
## Config Fields Used
|
||||||
Note: `scriptorium serve` and HTTP API exist, but they are not the initial integration path.
|
Via `pipeline.scriptorium.*` and stage-level artifact config:
|
||||||
|
- `binary`, `config_path`, `timeout`, `render_debug`
|
||||||
## 4. Command Selection Guidance
|
- artifact-level `prompt_id`, `profile_id`, `timeout`, `inputs`, `vars`, `output_path`
|
||||||
|
|
||||||
- Use `run` to generate an output artifact.
|
## External Adapters Used
|
||||||
- Use `render` to inspect the prepared prompt and effective settings without calling the LLM.
|
- Shared subprocess helper (`internal/adapters/subprocess`).
|
||||||
- Use `render --format json` when Narratio/tests need structured prepare output.
|
|
||||||
|
## State and Manifest Behavior
|
||||||
## 5. Recommended `run` Invocation Shape
|
- No direct manifest writes.
|
||||||
|
- Stage metadata records adapter outputs and command mode.
|
||||||
Production shape:
|
- Generated invocation YAML is written when requested.
|
||||||
|
|
||||||
```bash
|
## Skip and Resume Behavior
|
||||||
scriptorium run \
|
- Adapter has no skip/resume logic. Stage/runner controls execution.
|
||||||
--prompt <prompt_id> \
|
|
||||||
--input transcript=<processed-transcript-path> \
|
## Failure Behavior
|
||||||
--out <output-artifact-path>
|
- Request validation fails for missing binary/prompt/output, invalid timeout, invalid input/var names, or missing required API env var.
|
||||||
```
|
- Subprocess errors bubble with command context.
|
||||||
|
- `run` exit code 2 is treated as `ValidationFailed=true` and surfaced as error by calling stage.
|
||||||
Common optional additions:
|
- Successful subprocess still fails if output file is missing/empty.
|
||||||
|
|
||||||
- `--config <path>`: use a specific Scriptorium config file.
|
## Tests to Inspect Before Changing
|
||||||
- `--profile <profile_id>`: override prompt default profile.
|
- `internal/adapters/scriptorium/subprocess_test.go`
|
||||||
- `--var name=value` (repeatable): small metadata values.
|
- `internal/adapters/scriptorium/fake_test.go`
|
||||||
- `--input name=path` (repeatable): additional named artifacts.
|
- `internal/stage/analyze_test.go`
|
||||||
- `--timeout <duration>`: per-run timeout override.
|
- `internal/stage/trim_test.go`
|
||||||
- Runtime model override flags (`--llm-base-url`, `--model`, etc.) only for exceptional/operator-directed cases.
|
|
||||||
|
## Architectural Invariants
|
||||||
## 6. Recommended `render` Invocation Shape
|
- Both modes require explicit timeout > 0.
|
||||||
|
- Input/var maps are sorted into deterministic CLI argument order.
|
||||||
Human-readable debug shape:
|
- Run-mode validation failures are represented explicitly, not silently skipped.
|
||||||
|
|
||||||
```bash
|
|
||||||
scriptorium render \
|
|
||||||
--prompt <prompt_id> \
|
|
||||||
--input transcript=<processed-transcript-path> \
|
|
||||||
--format text
|
|
||||||
```
|
|
||||||
|
|
||||||
Structured debug/test shape:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
scriptorium render \
|
|
||||||
--prompt <prompt_id> \
|
|
||||||
--input transcript=<processed-transcript-path> \
|
|
||||||
--format json \
|
|
||||||
--out <render-debug-path>
|
|
||||||
```
|
|
||||||
|
|
||||||
`render` does **not** call the LLM, does **not** validate model output, and does **not** perform repair.
|
|
||||||
|
|
||||||
## 7. Inputs
|
|
||||||
|
|
||||||
- Pass inputs as repeated `--input name=path` flags.
|
|
||||||
- `name` must match the Prompt Definition input name.
|
|
||||||
- Prefer absolute paths, or paths relative to a working directory controlled by Narratio.
|
|
||||||
- Pass Audita output as the primary transcript input.
|
|
||||||
- Additional inputs may include glossary, previous recap, campaign notes, event logs, final state maps, or other prompt-specific artifacts.
|
|
||||||
- Scriptorium reads input files directly; Narratio does not need to inline file content for CLI use.
|
|
||||||
|
|
||||||
## 8. Variables
|
|
||||||
|
|
||||||
Use repeated `--var name=value` for small metadata values.
|
|
||||||
|
|
||||||
Typical examples:
|
|
||||||
|
|
||||||
- `session_date`
|
|
||||||
- `session_id`
|
|
||||||
- `campaign_name`
|
|
||||||
- `previous_session_id`
|
|
||||||
- `output_kind`
|
|
||||||
|
|
||||||
Large content belongs in input files, not `--var` values.
|
|
||||||
|
|
||||||
## 9. Prompt IDs and Output Artifact Types
|
|
||||||
|
|
||||||
Narratio should treat prompt IDs as configuration, not hardcoded business logic.
|
|
||||||
|
|
||||||
Narratio config may map stage/output names to prompt IDs, for example:
|
|
||||||
|
|
||||||
- session recap prompt
|
|
||||||
- structured event extraction prompt
|
|
||||||
- glossary suggestion prompt
|
|
||||||
- player-facing summary prompt
|
|
||||||
|
|
||||||
Prompt IDs used by Narratio should come from the deployed Scriptorium prompt library.
|
|
||||||
|
|
||||||
## 10. Profiles
|
|
||||||
|
|
||||||
- Prompts may declare `default_profile`.
|
|
||||||
- Narratio may omit `--profile` to use prompt default profile.
|
|
||||||
- Narratio may pass `--profile` to force profile selection.
|
|
||||||
- This enables environment/profile selection like `local-fast`, `local-quality`, `frontier`, `batch`, or test profiles.
|
|
||||||
- Profile names should generally be Narratio configuration values.
|
|
||||||
|
|
||||||
## 11. Runtime Overrides
|
|
||||||
|
|
||||||
Supported runtime override flags:
|
|
||||||
|
|
||||||
- `--llm-base-url`
|
|
||||||
- `--model`
|
|
||||||
- `--api-key-env`
|
|
||||||
- `--temperature`
|
|
||||||
- `--max-tokens`
|
|
||||||
- `--top-p`
|
|
||||||
- `--timeout`
|
|
||||||
|
|
||||||
Guidance:
|
|
||||||
|
|
||||||
- Keep normal model/runtime settings in Execution Profiles.
|
|
||||||
- Use runtime overrides only for explicit per-run exceptions, tests, or operator overrides.
|
|
||||||
- Never pass raw API keys on the command line.
|
|
||||||
- `--api-key-env` names an environment variable; Narratio must ensure that variable is set in subprocess environment.
|
|
||||||
|
|
||||||
## 12. Config Behavior
|
|
||||||
|
|
||||||
- Default config path: `/etc/scriptorium/config.yml`.
|
|
||||||
- `--config <PATH>` overrides default path.
|
|
||||||
- Missing default config is allowed by Scriptorium.
|
|
||||||
- If `--config` is provided explicitly, the file must exist and be valid.
|
|
||||||
- CLI flags override `config.yml`.
|
|
||||||
- `config.yml` overrides built-in application defaults.
|
|
||||||
|
|
||||||
Narratio can either:
|
|
||||||
|
|
||||||
- rely on system default config path, or
|
|
||||||
- carry an explicit config path and pass `--config`.
|
|
||||||
|
|
||||||
## 13. Environment Handling
|
|
||||||
|
|
||||||
Subprocess environment recommendations:
|
|
||||||
|
|
||||||
- Pass through required API-key environment variables referenced by `api_key_env`.
|
|
||||||
- Do not pass raw API keys as CLI arguments.
|
|
||||||
- Avoid logging full environment dumps.
|
|
||||||
- Capture stdout and stderr separately.
|
|
||||||
- Use a controlled working directory.
|
|
||||||
- Prefer absolute artifact paths.
|
|
||||||
|
|
||||||
## 14. Output Handling
|
|
||||||
|
|
||||||
For `scriptorium run`:
|
|
||||||
|
|
||||||
- Use `--out` when Narratio needs durable artifact files.
|
|
||||||
- Without `--out`, artifact content is written to stdout.
|
|
||||||
- Preferred orchestration pattern: always use `--out`, then treat the file as stage output artifact.
|
|
||||||
- Capture stderr for diagnostics.
|
|
||||||
|
|
||||||
For `scriptorium render`:
|
|
||||||
|
|
||||||
- Use `--out` to store render diagnostics.
|
|
||||||
- Use `--format json` when tests need to inspect selected profile, effective runtime settings, input hashes, prompt hash, and rendered messages.
|
|
||||||
|
|
||||||
## 15. Exit Status and Errors
|
|
||||||
|
|
||||||
Current CLI behavior (verified from implementation/tests):
|
|
||||||
|
|
||||||
- `0`: success.
|
|
||||||
- `1`: runtime/parse/config/load/render/generation/IO error.
|
|
||||||
- `2`: run completed but output validation failed (`ValidationFailed`).
|
|
||||||
|
|
||||||
Additional details:
|
|
||||||
|
|
||||||
- On `run`, output artifact write happens before exit code selection. If validation fails, artifact may still be written and exit code is `2`.
|
|
||||||
- `stderr` carries both errors and normal run summary output; non-empty stderr alone does not imply failure.
|
|
||||||
- `render` returns `0` on success and `1` on failures.
|
|
||||||
|
|
||||||
Narratio should treat non-zero exit codes as failed stage execution, but may record generated artifact paths if a run exited `2` and output file exists.
|
|
||||||
|
|
||||||
## 16. Recommended Narratio Integration Pattern
|
|
||||||
|
|
||||||
1. Build CLI args from Narratio stage configuration.
|
|
||||||
2. Use subprocess context cancellation/timeout.
|
|
||||||
3. Pass absolute input paths.
|
|
||||||
4. Pass `--out` to a session-scoped artifact path.
|
|
||||||
5. Add `--var` metadata values.
|
|
||||||
6. Optionally add `--config`.
|
|
||||||
7. Optionally add `--profile`.
|
|
||||||
8. Ensure required API-key env vars are present.
|
|
||||||
9. Run subprocess synchronously.
|
|
||||||
10. Capture stdout/stderr separately.
|
|
||||||
11. On success, store output artifact path and invocation metadata in stage artifacts.
|
|
||||||
12. On failure, store exit code and stderr diagnostics in stage status.
|
|
||||||
|
|
||||||
## 17. Suggested Narratio Configuration Shape
|
|
||||||
|
|
||||||
Illustrative `pipeline.yml` shape:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
scriptorium:
|
|
||||||
binary: scriptorium
|
|
||||||
config_path: /etc/scriptorium/config.yml
|
|
||||||
timeout: 10m
|
|
||||||
render_debug: false
|
|
||||||
artifacts:
|
|
||||||
session_recap:
|
|
||||||
enabled: true
|
|
||||||
prompt_id: dnd.session_recap
|
|
||||||
profile_id: local-quality # optional
|
|
||||||
output_path: artifacts/session_recap.md
|
|
||||||
timeout: 10m
|
|
||||||
render_debug: false # optional artifact override
|
|
||||||
inputs:
|
|
||||||
transcript:
|
|
||||||
source: trimmed_transcript
|
|
||||||
required: true
|
|
||||||
previous_recap:
|
|
||||||
source: previous_session_artifact
|
|
||||||
artifact: session_recap
|
|
||||||
path: "" # optional
|
|
||||||
required: false
|
|
||||||
vars:
|
|
||||||
session_id: true
|
|
||||||
session_date: true
|
|
||||||
campaign_name: true
|
|
||||||
previous_session_id: true
|
|
||||||
output_kind: session_recap
|
|
||||||
```
|
|
||||||
|
|
||||||
The key idea: map Narratio artifact names to prompt ID, optional profile, expected inputs, vars, and output destination.
|
|
||||||
|
|
||||||
## 18. Testing Strategy for Narratio Integration
|
|
||||||
|
|
||||||
- Use `scriptorium render --format json` to verify command construction without LLM calls.
|
|
||||||
- Use dedicated test prompt/profile libraries for integration tests.
|
|
||||||
- Use small fixture transcripts.
|
|
||||||
- Verify missing-input failure behavior.
|
|
||||||
- Verify prompt `default_profile` behavior.
|
|
||||||
- Verify explicit `--profile` override behavior.
|
|
||||||
- Verify `--config` behavior (default and explicit).
|
|
||||||
- Verify output file creation when `--out` is used.
|
|
||||||
- Verify stderr capture on failures.
|
|
||||||
- Avoid real API keys in tests.
|
|
||||||
|
|
||||||
## 19. Security and Privacy Notes
|
|
||||||
|
|
||||||
- Never pass raw API keys on command line.
|
|
||||||
- Do not log full rendered prompts by default; transcripts may contain sensitive content.
|
|
||||||
- Avoid logging prompt content unless explicit debug mode is enabled.
|
|
||||||
- Treat generated artifacts as potentially sensitive.
|
|
||||||
- Use session-scoped, access-controlled output paths.
|
|
||||||
- `api_key_env` names should come from environment management, not embedded secrets.
|
|
||||||
|
|
||||||
## 20. Initial D&D Artifact Generation Examples
|
|
||||||
|
|
||||||
These are examples only. Use prompt IDs from the deployed prompt library.
|
|
||||||
|
|
||||||
Session recap:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
scriptorium run \
|
|
||||||
--prompt dnd.session_recap \
|
|
||||||
--input transcript=/work/campaign-7/session-42/transcript.polished.md \
|
|
||||||
--input glossary=/work/campaign-7/session-42/glossary.yml \
|
|
||||||
--out /work/campaign-7/session-42/artifacts/session_recap.md
|
|
||||||
```
|
|
||||||
|
|
||||||
Structured events:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
scriptorium run \
|
|
||||||
--prompt dnd.structured_events \
|
|
||||||
--input transcript=/work/campaign-7/session-42/transcript.polished.md \
|
|
||||||
--out /work/campaign-7/session-42/artifacts/structured_events.json
|
|
||||||
```
|
|
||||||
|
|
||||||
Glossary suggestions:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
scriptorium run \
|
|
||||||
--prompt dnd.glossary_suggestions \
|
|
||||||
--input transcript=/work/campaign-7/session-42/transcript.polished.md \
|
|
||||||
--input previous_recap=/work/campaign-7/session-41/artifacts/session_recap.md \
|
|
||||||
--out /work/campaign-7/session-42/artifacts/glossary_suggestions.md
|
|
||||||
```
|
|
||||||
|
|
||||||
Player-facing summary:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
scriptorium run \
|
|
||||||
--prompt dnd.player_summary \
|
|
||||||
--input transcript=/work/campaign-7/session-42/transcript.polished.md \
|
|
||||||
--input structured_events=/work/campaign-7/session-42/artifacts/structured_events.json \
|
|
||||||
--out /work/campaign-7/session-42/artifacts/player_summary.md
|
|
||||||
```
|
|
||||||
|
|
||||||
## 21. Non-Goals
|
|
||||||
|
|
||||||
Initial Narratio integration should not:
|
|
||||||
|
|
||||||
- call Scriptorium internal Go packages
|
|
||||||
- use HTTP API as the primary path
|
|
||||||
- expect Scriptorium to read S3 refs directly
|
|
||||||
- make Scriptorium responsible for Narratio stage state
|
|
||||||
- make Scriptorium responsible for notification
|
|
||||||
- require Scriptorium to understand D&D workflow semantics beyond prompt definitions
|
|
||||||
|
|
||||||
## 22. Future Extension Notes
|
|
||||||
|
|
||||||
Possible later extensions:
|
|
||||||
|
|
||||||
- HTTP API integration
|
|
||||||
- S3 artifact references if Scriptorium adds S3 reader support
|
|
||||||
- richer render diagnostics and policy controls
|
|
||||||
- token budgeting/prompt-size checks
|
|
||||||
- batch execution if Scriptorium later adds batch support
|
|
||||||
|
|||||||
@@ -1,403 +1,60 @@
|
|||||||
# seriatim
|
# Integration: seriatim
|
||||||
|
|
||||||
`seriatim` merges per-speaker WhisperX-style JSON transcripts into a single JSON transcript that preserves speaker identity and chronological order.
|
## Purpose
|
||||||
|
Define Narratio's adapter contract for merge, normalize, and trim subprocess invocations of Seriatim.
|
||||||
The current implementation supports the `merge` command. It reads one or more input JSON files, optionally maps each input file to a canonical speaker using `speakers.yml`, sorts all segments by timestamp, detects and resolves overlaps when word-level timing is available, assigns consecutive numeric `id` values, and writes a merged JSON artifact.
|
|
||||||
|
## Inputs and Outputs
|
||||||
## Usage
|
Inputs:
|
||||||
|
- `MergeRequest`: raw/normalized transcript inputs, output path, optional report, speaker/autocorrect paths, logs/config
|
||||||
Run from source:
|
- `NormalizeRequest`: input transcript, output path, schema, optional report, timeout/log/config
|
||||||
|
- `TrimRequest`: input transcript, output path, keep selector, timeout/log/config
|
||||||
```sh
|
|
||||||
go run ./cmd/seriatim merge \
|
Outputs:
|
||||||
--input-file samples/raw/2026-04-19-Eric_Rakestraw.json \
|
- `MergeResult`, `NormalizeResult`, `TrimResult` with output paths, logs/config paths, exit code, duration, binary provenance, and metadata.
|
||||||
--input-file samples/raw/2026-04-19-Mike_Brown.json \
|
|
||||||
--output-file merged.json
|
## Boundaries
|
||||||
```
|
Owns:
|
||||||
|
- Validated deterministic CLI invocation construction
|
||||||
Optional report output:
|
- Optional env tuning propagation for merge
|
||||||
|
- Invocation config file emission
|
||||||
```sh
|
- JSON output validation
|
||||||
go run ./cmd/seriatim merge \
|
|
||||||
--input-file eric.json \
|
Does not own:
|
||||||
--input-file mike.json \
|
- Transcript input selection/promotion logic (stage-owned)
|
||||||
--output-file merged.json \
|
- Bounds computation (scriptorium/trim-stage-owned)
|
||||||
--report-file report.json
|
|
||||||
```
|
## Config Fields Used
|
||||||
|
Via `pipeline.seriatim.*` mapped in app/stage wiring:
|
||||||
## CLI
|
- `binary`, `timeout`, `output_schema`, `coalesce_gap`, `report`
|
||||||
|
- `env.overlap_word_run_gap`
|
||||||
```text
|
- `env.overlap_word_run_reorder_window`
|
||||||
seriatim merge [flags]
|
- `env.backchannel_max_duration`
|
||||||
```
|
- `env.filler_max_duration`
|
||||||
|
|
||||||
Global flags:
|
## External Adapters Used
|
||||||
|
- Shared subprocess helper (`internal/adapters/subprocess`).
|
||||||
| Flag | Description |
|
|
||||||
| --- | --- |
|
## State and Manifest Behavior
|
||||||
| `--help` | Show command help. |
|
- No direct manifest writes.
|
||||||
| `--version` | Show application version. Local builds default to `dev`; release builds inject the release version. |
|
- Stage metadata consumes adapter result fields and preserves generated config/log references.
|
||||||
|
|
||||||
`merge` flags:
|
## Skip and Resume Behavior
|
||||||
|
- Adapter has no skip/resume logic. Runner controls stage execution.
|
||||||
| Flag | Required | Default | Description |
|
|
||||||
| --- | --- | --- | --- |
|
## Failure Behavior
|
||||||
| `--input-file` | Yes | none | Input transcript JSON file. Repeat once per speaker/input file. |
|
- Constructor fails for invalid binary/timeout/output-schema/coalesce-gap.
|
||||||
| `--output-file` | Yes | none | Merged transcript JSON output path. |
|
- Merge fails on missing output path/inputs/report path (if enabled), subprocess errors, invalid merged output JSON, invalid report JSON.
|
||||||
| `--report-file` | No | none | Optional report JSON output path. |
|
- Normalize fails on missing input/output, invalid schema, subprocess errors, invalid normalized output JSON shape, invalid report JSON.
|
||||||
| `--speakers` | No | none | Speaker map YAML file. When omitted, input file basenames are used as speaker labels. |
|
- Trim fails on missing input/output/keep selector, subprocess errors, invalid trimmed output JSON shape.
|
||||||
| `--autocorrect` | No | none | Autocorrect rules YAML file. When omitted, the default `autocorrect` module leaves text unchanged. |
|
|
||||||
| `--input-reader` | No | `json-files` | Input reader module. |
|
## Tests to Inspect Before Changing
|
||||||
| `--output-modules` | No | `json` | Comma-separated output modules. |
|
- `internal/adapters/seriatim/subprocess_test.go`
|
||||||
| `--output-schema` | No | `seriatim-intermediate` | JSON output contract. Allowed values are `seriatim-minimal`, `seriatim-intermediate`, and `seriatim-full`. If omitted, the runtime default is used; consumers that depend on a specific shape should set this explicitly. |
|
- `internal/adapters/seriatim/fake_test.go`
|
||||||
| `--preprocessing-modules` | No | `validate-raw,normalize-speakers,trim-text` | Comma-separated preprocessing modules, evaluated in order. |
|
- `internal/stage/merge_test.go`
|
||||||
| `--postprocessing-modules` | No | `detect-overlaps,resolve-overlaps,backchannel,filler,resolve-danglers,coalesce,detect-overlaps,autocorrect,assign-ids,validate-output` | Comma-separated postprocessing modules, evaluated in order. |
|
- `internal/stage/normalize_test.go`
|
||||||
| `--coalesce-gap` | No | `3.0` | Maximum same-speaker gap in seconds for `coalesce`; also used as the `resolve-overlaps` context window. Must be a non-negative float. |
|
- `internal/stage/trim_test.go`
|
||||||
|
|
||||||
Environment variables:
|
## Architectural Invariants
|
||||||
|
- Supported output schemas are limited to `seriatim-minimal`, `seriatim-intermediate`, `seriatim-full`.
|
||||||
| Environment Variable | Default | Description |
|
- Normalize/trim outputs must include `segments` arrays.
|
||||||
| --- | --- | --- |
|
- Merge/normalize/trim all route through deterministic subprocess invocation.
|
||||||
| `SERIATIM_OUTPUT_SCHEMA` | `seriatim-intermediate` | Output schema used when `--output-schema` is not explicitly provided. Allowed values are `seriatim-minimal`, `seriatim-intermediate`, and `seriatim-full`. The CLI flag takes precedence. |
|
|
||||||
| `SERIATIM_OVERLAP_WORD_RUN_GAP` | `1.0` | Maximum gap in seconds between adjacent timed words when `resolve-overlaps` builds word-run replacement segments. Must be a positive float. |
|
|
||||||
| `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW` | `1.0` | Near-start window in seconds for ordering replacement word runs shortest-first. Must be a positive float. |
|
|
||||||
| `SERIATIM_BACKCHANNEL_MAX_DURATION` | `2.0` | Maximum duration in seconds for `backchannel` classification. Must be a positive float. |
|
|
||||||
| `SERIATIM_FILLER_MAX_DURATION` | `1.25` | Maximum duration in seconds for `filler` classification. Must be a positive float. |
|
|
||||||
|
|
||||||
## Input JSON Format
|
|
||||||
|
|
||||||
Each input file must be valid JSON with a top-level `segments` array. The current parser accepts the WhisperX segment subset needed for merging:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"segments": [
|
|
||||||
{
|
|
||||||
"start": 1.25,
|
|
||||||
"end": 3.5,
|
|
||||||
"text": "Hello there.",
|
|
||||||
"words": [
|
|
||||||
{"word": "Hello", "start": 1.25, "end": 1.55, "score": 0.98},
|
|
||||||
{"word": "there.", "start": 1.7, "end": 2.0}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
Required segment fields:
|
|
||||||
|
|
||||||
- `start`: number, must be `>= 0`.
|
|
||||||
- `end`: number, must be `>= start`.
|
|
||||||
- `text`: string.
|
|
||||||
|
|
||||||
Optional word fields:
|
|
||||||
|
|
||||||
- `words`: array of word timing objects.
|
|
||||||
- `words[].word`: string.
|
|
||||||
- `words[].start`: optional number, must be `>= 0` when present.
|
|
||||||
- `words[].end`: optional number, must be `>= start` when present with `start`.
|
|
||||||
- `words[].score`: optional number.
|
|
||||||
- `words[].speaker`: optional raw speaker label string.
|
|
||||||
|
|
||||||
Word-level timing is preserved internally for overlap resolution. If a word is missing `start` or `end`, seriatim keeps the word text, emits a warning in the optional report, and does not use that word as a timing anchor. Word timing is not emitted in the final JSON artifact.
|
|
||||||
|
|
||||||
## Speaker Map Format
|
|
||||||
|
|
||||||
`speakers.yml` maps input files to canonical speaker names using ordered substring rules:
|
|
||||||
|
|
||||||
This file is optional. If `--speakers` is omitted, `seriatim` uses each input file basename as the segment speaker label.
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
match:
|
|
||||||
- speaker: "Eric Rakestraw"
|
|
||||||
match:
|
|
||||||
- "Eric_Rakestraw"
|
|
||||||
- "Eric"
|
|
||||||
|
|
||||||
- speaker: "Mike Brown"
|
|
||||||
match:
|
|
||||||
- "Mike_Brown"
|
|
||||||
- "mb"
|
|
||||||
```
|
|
||||||
|
|
||||||
For each `--input-file`, `seriatim` takes the file basename and evaluates the rules in order. The first rule with a matching substring wins, and no later rules are evaluated.
|
|
||||||
|
|
||||||
For example, this input:
|
|
||||||
|
|
||||||
```text
|
|
||||||
samples/raw/2026-04-19-Eric_Rakestraw.json
|
|
||||||
```
|
|
||||||
|
|
||||||
matches this rule because the basename contains `Eric_Rakestraw`:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
- speaker: "Eric Rakestraw"
|
|
||||||
match:
|
|
||||||
- "Eric_Rakestraw"
|
|
||||||
```
|
|
||||||
|
|
||||||
Important details:
|
|
||||||
|
|
||||||
- Matching is against the input file basename, not the full path.
|
|
||||||
- Matching is case-insensitive.
|
|
||||||
- Rules are evaluated from first to last.
|
|
||||||
- Each rule must have a non-empty `speaker`.
|
|
||||||
- Each rule must have at least one non-empty `match` string.
|
|
||||||
- Duplicate speaker names are invalid.
|
|
||||||
- Every input file must match at least one rule or the command fails.
|
|
||||||
|
|
||||||
Deprecated old format:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
inputs:
|
|
||||||
eric.json:
|
|
||||||
speaker: "Eric Rakestraw"
|
|
||||||
```
|
|
||||||
|
|
||||||
The old `inputs:` direct mapping format is no longer supported.
|
|
||||||
|
|
||||||
## Output JSON Format
|
|
||||||
|
|
||||||
`--output-modules json` controls the writer. `--output-schema` controls the JSON contract that writer serializes.
|
|
||||||
|
|
||||||
The named schemas are stable public contracts. If a consumer depends on a specific shape, it should request that schema explicitly at runtime. The runtime default selection may change in a future release.
|
|
||||||
|
|
||||||
The `seriatim-intermediate` schema is the current default selection when neither `--output-schema` nor `SERIATIM_OUTPUT_SCHEMA` is set. It stays close to the minimal schema, but adds optional `categories` on each segment:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"metadata": {
|
|
||||||
"application": "seriatim",
|
|
||||||
"version": "dev",
|
|
||||||
"output_schema": "seriatim-intermediate"
|
|
||||||
},
|
|
||||||
"segments": [
|
|
||||||
{
|
|
||||||
"id": 1,
|
|
||||||
"start": 1.25,
|
|
||||||
"end": 3.5,
|
|
||||||
"speaker": "Eric Rakestraw",
|
|
||||||
"text": "Hello there.",
|
|
||||||
"categories": ["backchannel"]
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
The `seriatim-full` schema uses the full seriatim envelope:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"metadata": {
|
|
||||||
"application": "seriatim",
|
|
||||||
"version": "dev",
|
|
||||||
"input_reader": "json-files",
|
|
||||||
"input_files": ["eric.json", "mike.json"],
|
|
||||||
"preprocessing_modules": ["validate-raw", "normalize-speakers", "trim-text"],
|
|
||||||
"postprocessing_modules": ["detect-overlaps", "resolve-overlaps", "backchannel", "filler", "resolve-danglers", "coalesce", "detect-overlaps", "autocorrect", "assign-ids", "validate-output"],
|
|
||||||
"output_modules": ["json"]
|
|
||||||
},
|
|
||||||
"segments": [
|
|
||||||
{
|
|
||||||
"id": 1,
|
|
||||||
"source": "eric.json",
|
|
||||||
"source_segment_index": 0,
|
|
||||||
"speaker": "Eric Rakestraw",
|
|
||||||
"start": 1.25,
|
|
||||||
"end": 3.5,
|
|
||||||
"text": "Hello there.",
|
|
||||||
"overlap_group_id": 1
|
|
||||||
},
|
|
||||||
{
|
|
||||||
"id": 2,
|
|
||||||
"source": "eric.json",
|
|
||||||
"source_ref": "word-run:1:1:1",
|
|
||||||
"derived_from": ["eric.json#0"],
|
|
||||||
"speaker": "Eric Rakestraw",
|
|
||||||
"start": 2.0,
|
|
||||||
"end": 2.5,
|
|
||||||
"text": "Resolved word run",
|
|
||||||
"categories": ["backchannel"]
|
|
||||||
}
|
|
||||||
],
|
|
||||||
"overlap_groups": [
|
|
||||||
{
|
|
||||||
"id": 1,
|
|
||||||
"start": 1.25,
|
|
||||||
"end": 4.0,
|
|
||||||
"segments": ["eric.json#0", "mike.json#0"],
|
|
||||||
"speakers": ["Eric Rakestraw", "Mike Brown"],
|
|
||||||
"class": "unknown",
|
|
||||||
"resolution": "unresolved"
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
The `seriatim-minimal` schema emits minimal metadata and compact ordered segments:
|
|
||||||
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"metadata": {
|
|
||||||
"application": "seriatim",
|
|
||||||
"version": "dev",
|
|
||||||
"output_schema": "seriatim-minimal"
|
|
||||||
},
|
|
||||||
"segments": [
|
|
||||||
{
|
|
||||||
"id": 1,
|
|
||||||
"start": 1.25,
|
|
||||||
"end": 3.5,
|
|
||||||
"speaker": "Eric Rakestraw",
|
|
||||||
"text": "Hello there."
|
|
||||||
}
|
|
||||||
]
|
|
||||||
}
|
|
||||||
```
|
|
||||||
|
|
||||||
Minimal output intentionally omits categories, overlap groups, source/provenance fields, and pipeline configuration metadata.
|
|
||||||
|
|
||||||
Intermediate output intentionally omits overlap groups and source/provenance fields, but keeps optional `categories` and minimal metadata.
|
|
||||||
|
|
||||||
Segments are sorted deterministically by:
|
|
||||||
|
|
||||||
```text
|
|
||||||
(start, end, source, source_segment_index/source_ref, speaker)
|
|
||||||
```
|
|
||||||
|
|
||||||
Final segment IDs are assigned after sorting and start at `1`.
|
|
||||||
|
|
||||||
The public Go output contract is available from:
|
|
||||||
|
|
||||||
```go
|
|
||||||
import "gitea.maximumdirect.net/eric/seriatim/schema"
|
|
||||||
```
|
|
||||||
|
|
||||||
The same package embeds machine-readable JSON Schemas in `schema/full-output.schema.json`, `schema/intermediate-output.schema.json`, and `schema/minimal-output.schema.json`. The default `validate-output` postprocessor validates the selected output shape and verifies final segment IDs are present, sequential, and start at `1`.
|
|
||||||
|
|
||||||
## Overlap Detection
|
|
||||||
|
|
||||||
The default postprocessing pipeline detects overlapping segment groups.
|
|
||||||
|
|
||||||
Overlap behavior:
|
|
||||||
|
|
||||||
- A strict timing overlap is required: `next.start < current_group_end`.
|
|
||||||
- Segments that only touch at a boundary are not grouped.
|
|
||||||
- Groups require at least two distinct speakers.
|
|
||||||
- Transitive overlaps are grouped together.
|
|
||||||
- Segments in detected groups receive `overlap_group_id`.
|
|
||||||
- `overlap_groups[].segments` contains stable references in `source#source_segment_index` format.
|
|
||||||
- `class` is currently `unknown`.
|
|
||||||
- `resolution` is `unresolved` until `resolve-overlaps` replaces the group.
|
|
||||||
|
|
||||||
## Overlap Resolution
|
|
||||||
|
|
||||||
The default postprocessing pipeline runs `detect-overlaps`, then `resolve-overlaps`, then `backchannel`, then `filler`, then `resolve-danglers`, then `coalesce`, then a second `detect-overlaps` pass.
|
|
||||||
|
|
||||||
For each detected overlap group, `resolve-overlaps` uses preserved WhisperX word timing to build smaller word-run replacement segments:
|
|
||||||
|
|
||||||
- The resolution window expands the detected overlap group by `--coalesce-gap` seconds on both sides.
|
|
||||||
- Nearby same-speaker context segments are included when they intersect the expanded window and their start or end is within `--coalesce-gap` of the original overlap boundary.
|
|
||||||
- Once a segment is selected for replacement, all timed words from that segment participate in word-run construction; the window controls segment selection, not per-word clipping.
|
|
||||||
- Context segments that are part of another detected overlap group are not pulled into the current group.
|
|
||||||
- Untimed words are included in replacement text in original word order when nearby timed words create a replacement run.
|
|
||||||
- Untimed words do not affect replacement segment start/end times or word-run gap splitting.
|
|
||||||
- Words for the same speaker are merged into one run when the gap between adjacent words is no greater than `SERIATIM_OVERLAP_WORD_RUN_GAP`.
|
|
||||||
- The default word-run gap is `1.0` seconds.
|
|
||||||
- Set `SERIATIM_OVERLAP_WORD_RUN_GAP` to a positive number of seconds to override the default.
|
|
||||||
- Near-start replacement word runs are reordered so shorter segments come first when adjacent starts are within `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW`.
|
|
||||||
- The default word-run reorder window is `1.0` seconds.
|
|
||||||
- Set `SERIATIM_OVERLAP_WORD_RUN_REORDER_WINDOW` to a positive number of seconds to override the default.
|
|
||||||
- Replacement segment text is built by joining word text with single spaces.
|
|
||||||
- Replacement segments include `source_ref` and `derived_from`.
|
|
||||||
- Replacement segments omit `source_segment_index` because they are derived from one or more original segments.
|
|
||||||
- Resolved overlap groups are removed before the second detection pass.
|
|
||||||
- Replacement segments are left without `overlap_group_id` until the second detection pass annotates any remaining overlap.
|
|
||||||
- If a speaker has no usable word timing in a group, that speaker's original segment is kept.
|
|
||||||
- If no speakers in a group have usable word timing, the original group and annotations remain unchanged.
|
|
||||||
|
|
||||||
## Backchannels
|
|
||||||
|
|
||||||
The default pipeline runs `backchannel` before `coalesce`. It tags short acknowledgement segments with:
|
|
||||||
|
|
||||||
```json
|
|
||||||
"categories": ["backchannel"]
|
|
||||||
```
|
|
||||||
|
|
||||||
Backchannel matching is case-insensitive, ignores punctuation for matching and word-count purposes, trims surrounding whitespace, and requires a matching acknowledgement phrase, no more than three whitespace-delimited words, and duration no greater than `SERIATIM_BACKCHANNEL_MAX_DURATION` seconds. The default maximum duration is `2.0` seconds.
|
|
||||||
|
|
||||||
## Fillers
|
|
||||||
|
|
||||||
The default pipeline runs `filler` after `backchannel` and before `coalesce`. It tags short filler utterances with:
|
|
||||||
|
|
||||||
```json
|
|
||||||
"categories": ["filler"]
|
|
||||||
```
|
|
||||||
|
|
||||||
Filler matching is case-insensitive, ignores punctuation for matching and word-count purposes, trims surrounding whitespace, and requires only filler tokens such as `um`, `uh`, `er`, `erm`, `ah`, `eh`, `hmm`, `mm`, or repeated combinations of those tokens. Matching segments must contain no more than three whitespace-delimited words and have duration no greater than `SERIATIM_FILLER_MAX_DURATION` seconds. The default maximum duration is `1.25` seconds.
|
|
||||||
|
|
||||||
## Dangler Resolution
|
|
||||||
|
|
||||||
The default pipeline runs `resolve-danglers` before `coalesce` and before the second overlap detection pass. It repairs short derived fragments when they share provenance with a nearby segment:
|
|
||||||
|
|
||||||
- Dangling-end fragments have no more than two words and end in punctuation.
|
|
||||||
- Dangling-start fragments have no more than two words.
|
|
||||||
- Matching uses same-speaker segments with any shared `derived_from` value.
|
|
||||||
- Merged segments use `source_ref` values such as `resolve-danglers:1`, keep the target segment's transcript position, and union `derived_from`.
|
|
||||||
|
|
||||||
## Coalescing
|
|
||||||
|
|
||||||
The default pipeline runs `coalesce` after `resolve-danglers` and before the second overlap detection pass. It merges adjacent same-speaker segments in the transcript's current order when `next.start - current.end <= --coalesce-gap`.
|
|
||||||
|
|
||||||
Coalesced segments use `source_ref` values such as `coalesce:1`, include `derived_from`, and omit `source_segment_index`.
|
|
||||||
|
|
||||||
Different-speaker backchannel and filler segments do not block coalescing of surrounding same-speaker segments. Same-speaker backchannel and filler segments are merged normally when they are within `--coalesce-gap`. When same-speaker segments are coalesced, any `backchannel` or `filler` category from the merged inputs is dropped from the coalesced segment.
|
|
||||||
|
|
||||||
## Autocorrect
|
|
||||||
|
|
||||||
Autocorrect is included in the default postprocessing pipeline. If `--autocorrect` is omitted, the module leaves transcript text unchanged and records a skip event in the optional report.
|
|
||||||
|
|
||||||
Enable corrections by passing `--autocorrect`:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
go run ./cmd/seriatim merge \
|
|
||||||
--input-file input.json \
|
|
||||||
--autocorrect autocorrect.yml \
|
|
||||||
--output-file merged.json
|
|
||||||
```
|
|
||||||
|
|
||||||
`autocorrect.yml` format:
|
|
||||||
|
|
||||||
```yaml
|
|
||||||
autocorrect:
|
|
||||||
- target: "Hrank"
|
|
||||||
match:
|
|
||||||
- "hrank"
|
|
||||||
- "Frank"
|
|
||||||
|
|
||||||
- target: "Mike Brown"
|
|
||||||
match:
|
|
||||||
- "Mike Pat"
|
|
||||||
```
|
|
||||||
|
|
||||||
Matching behavior:
|
|
||||||
|
|
||||||
- Matching is case-sensitive.
|
|
||||||
- Matches apply only to whole tokens, not substrings inside larger words.
|
|
||||||
- Punctuation and whitespace can surround a match.
|
|
||||||
- Multi-word and hyphenated matches are supported.
|
|
||||||
- Duplicate match strings are invalid, including duplicates across separate rules.
|
|
||||||
|
|
||||||
## Current Limitations
|
|
||||||
|
|
||||||
- Only JSON input is supported.
|
|
||||||
- Overlap resolution depends on WhisperX word timing; groups without usable word timing remain unresolved.
|
|
||||||
- Alternate output formats are not implemented yet.
|
|
||||||
|
|
||||||
## Release Builds
|
|
||||||
|
|
||||||
Local builds record version metadata as `dev`. Release builds should inject the release version with `ldflags`:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
go build -ldflags "-X gitea.maximumdirect.net/eric/seriatim/internal/buildinfo.Version=v1.0.0" ./cmd/seriatim
|
|
||||||
```
|
|
||||||
|
|||||||
28
docs/internal/README.md
Normal file
28
docs/internal/README.md
Normal file
@@ -0,0 +1,28 @@
|
|||||||
|
# Internal Documentation Index
|
||||||
|
|
||||||
|
## Audience
|
||||||
|
Developers and LLM coding agents changing Narratio internals.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
Implementation-accurate contracts for workspace/state, manifests, stages, artifact resolution, and adapter boundaries.
|
||||||
|
|
||||||
|
## Component Docs
|
||||||
|
- `adapters.md`: external adapter map, runtime wiring, and boundary ownership.
|
||||||
|
- `storage.md`: remote storage backend contracts and object-store invariants.
|
||||||
|
- `manifest.md`: session/run manifest schemas, lifecycle transitions, and persistence semantics.
|
||||||
|
- `artifacts.md`: built-in artifact registry, runtime artifact catalog, and source-resolution behavior.
|
||||||
|
- `workspace.md`: local state model, manifests, run-local layout, promotion, and cleanup invariants.
|
||||||
|
- `stage-prepare.md`: input materialization and provenance capture.
|
||||||
|
- `stage-transcribe.md`: WhisperX transcript generation.
|
||||||
|
- `stage-merge.md`: Seriatim normalization + merge.
|
||||||
|
- `stage-polish.md`: Audita transcript polishing.
|
||||||
|
- `stage-normalize.md`: post-polish normalization.
|
||||||
|
- `stage-trim.md`: bounds-driven transcript trimming.
|
||||||
|
- `stage-analyze.md`: dependency-ordered Scriptorium artifact generation for selected configured artifacts.
|
||||||
|
- `stage-archive.md`: archive upload and current-pointer publish contract.
|
||||||
|
|
||||||
|
## External Integration Notes
|
||||||
|
- `../integrations/README.md`: canonical location for external integration contracts (`audita.md`, `seriatim.md`, `scriptorium.md`).
|
||||||
|
|
||||||
|
## Canonical Owner
|
||||||
|
`docs/internal/` is the canonical home for implemented internals per `docs/documentation/policy.md`.
|
||||||
79
docs/internal/adapters.md
Normal file
79
docs/internal/adapters.md
Normal file
@@ -0,0 +1,79 @@
|
|||||||
|
# Internal: Adapters
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Describe the external adapter boundaries used by Narratio stages and app orchestration, including default runtime wiring.
|
||||||
|
|
||||||
|
## Inputs and outputs
|
||||||
|
Inputs:
|
||||||
|
- Stage requests passed through adapter interfaces (for example transcription, merge/normalize/trim, polish, artifact generation, object-store operations, notifications).
|
||||||
|
- Resolved config values used to construct default adapters.
|
||||||
|
|
||||||
|
Outputs:
|
||||||
|
- Adapter-specific result structs (paths, metadata, status/attempt info, duration/exit details).
|
||||||
|
- Adapter errors returned to stage/app orchestration.
|
||||||
|
|
||||||
|
## Boundaries
|
||||||
|
Owns:
|
||||||
|
- Transport/process/SDK details at system boundaries (`HTTP`, subprocess CLI invocation, AWS SDK calls).
|
||||||
|
- Request/response contracts in `internal/adapters/*` packages.
|
||||||
|
|
||||||
|
Does not own:
|
||||||
|
- Stage sequencing, skip/force/resume decisions.
|
||||||
|
- Manifest transition logic.
|
||||||
|
- Canonical workspace path policy.
|
||||||
|
|
||||||
|
## Config fields used
|
||||||
|
Default wiring and adapter calls consume:
|
||||||
|
- `pipeline.whisperx.*`
|
||||||
|
- `pipeline.seriatim.*`
|
||||||
|
- `pipeline.audita.*`
|
||||||
|
- `pipeline.scriptorium.*`
|
||||||
|
- `pipeline.storage.*` and `pipeline.archive.*` (object-store construction/gating)
|
||||||
|
- `pipeline.notification.*` (sender boundary exists; placeholder behavior today)
|
||||||
|
|
||||||
|
## External adapters used
|
||||||
|
Runtime env boundary fields (`internal/stage.Env`):
|
||||||
|
- `whisperx.Client`
|
||||||
|
- `seriatim.Runner`
|
||||||
|
- `audita.Runner`
|
||||||
|
- `scriptorium.Runner`
|
||||||
|
- `storage.ObjectStore`
|
||||||
|
- `notify.Sender`
|
||||||
|
- `analyzer.Runner`
|
||||||
|
|
||||||
|
Current execution usage:
|
||||||
|
- Actively used by implemented stages: `WhisperX`, `Seriatim`, `Audita`, `Scriptorium`, `ObjectStore`, `Notifier`.
|
||||||
|
- Present but not used by implemented stage set: `Analyzer`, legacy `storage.Backend`.
|
||||||
|
|
||||||
|
Default construction in app runner:
|
||||||
|
- Auto-constructed when not injected: WhisperX HTTP client, Seriatim subprocess runner, Audita subprocess runner, Scriptorium subprocess runner, object store (only when needed), and `notify.NoopSender`.
|
||||||
|
- Callers can inject test/fake implementations through `app.RunOptions.Env`.
|
||||||
|
|
||||||
|
## State and manifest behavior
|
||||||
|
- Adapters do not directly mutate session/run manifests.
|
||||||
|
- Stages and runner own manifest writes and stage status transitions.
|
||||||
|
- Adapter outputs are persisted indirectly through stage result mapping (outputs/logs/generated configs/metadata).
|
||||||
|
|
||||||
|
## Skip and resume behavior
|
||||||
|
- No adapter-level skip/resume semantics.
|
||||||
|
- Skip/resume/force behavior is decided by app runner using manifest stage state.
|
||||||
|
|
||||||
|
## Failure behavior
|
||||||
|
- Adapter constructors validate config-derived values and fail early on invalid required inputs.
|
||||||
|
- Adapter run-time failures are returned to stage code with boundary context and are recorded as stage failures by runner logic.
|
||||||
|
- Subprocess adapters preserve stdout/stderr and generated-config paths to aid diagnosis.
|
||||||
|
|
||||||
|
## Tests to inspect before changing
|
||||||
|
- `internal/adapters/whisperx/http_test.go`
|
||||||
|
- `internal/adapters/seriatim/subprocess_test.go`
|
||||||
|
- `internal/adapters/audita/subprocess_test.go`
|
||||||
|
- `internal/adapters/scriptorium/subprocess_test.go`
|
||||||
|
- `internal/adapters/storage/*_test.go`
|
||||||
|
- `internal/adapters/notify/fake_test.go`
|
||||||
|
- `internal/adapters/analyzer/fake_test.go`
|
||||||
|
- `internal/app/runner_test.go`
|
||||||
|
|
||||||
|
## Architectural invariants
|
||||||
|
- Stage code depends on adapter interfaces, not transport-specific implementation types.
|
||||||
|
- External SDK-specific types remain inside adapter implementations.
|
||||||
|
- Default app wiring must remain deterministic and overrideable via injected env dependencies.
|
||||||
87
docs/internal/artifacts.md
Normal file
87
docs/internal/artifacts.md
Normal file
@@ -0,0 +1,87 @@
|
|||||||
|
# Internal: Artifacts
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Define Narratio's artifact identity and resolution model for built-in transcript/bounds artifacts and runtime-configured analyze artifacts.
|
||||||
|
|
||||||
|
## Inputs and outputs
|
||||||
|
Inputs:
|
||||||
|
- artifact sources from config/runtime (`pipeline.scriptorium.artifacts.*.inputs.*.source`)
|
||||||
|
- session paths and optional session manifest stage outputs
|
||||||
|
- runtime artifact catalog state for configured artifact sources
|
||||||
|
|
||||||
|
Outputs:
|
||||||
|
- resolved local artifact path and provenance (`ResolvedSessionArtifact`)
|
||||||
|
- runtime catalog entries for planned/executable/available artifacts
|
||||||
|
- validation errors for unsupported, missing, or invalid artifact sources
|
||||||
|
|
||||||
|
## Boundaries
|
||||||
|
Owns:
|
||||||
|
- built-in artifact registry and content validation rules
|
||||||
|
- runtime artifact catalog for configured artifact source IDs
|
||||||
|
- source resolution behavior for built-in and configured artifact sources
|
||||||
|
|
||||||
|
Does not own:
|
||||||
|
- artifact generation (stages produce files)
|
||||||
|
- manifest transition policy
|
||||||
|
- archive promotion behavior
|
||||||
|
|
||||||
|
## Config fields used
|
||||||
|
- `pipeline.scriptorium.artifacts.<name>.enabled`
|
||||||
|
- `pipeline.scriptorium.artifacts.<name>.output_path`
|
||||||
|
- `pipeline.scriptorium.artifacts.<name>.inputs.<key>.source`
|
||||||
|
|
||||||
|
## External adapters used
|
||||||
|
- none
|
||||||
|
|
||||||
|
## State and manifest behavior
|
||||||
|
Built-in registry entries:
|
||||||
|
|
||||||
|
| Artifact ID | Canonical file | Producer stage | Output kind |
|
||||||
|
| --- | --- | --- | --- |
|
||||||
|
| `narratio.transcript.merged` | `transcripts/merged.json` | `merge` | `transcript_merged` |
|
||||||
|
| `narratio.transcript.polished` | `transcripts/processed.json` | `polish` | `transcript_processed` |
|
||||||
|
| `narratio.transcript.full` | `transcripts/normalized.json` | `normalize` | `transcript_normalized` |
|
||||||
|
| `narratio.transcript.trimmed` | `transcripts/trimmed.json` | `trim` | `transcript_trimmed` |
|
||||||
|
| `narratio.bounds.session` | `artifacts/session_bounds.json` | `trim` | `session_bounds` |
|
||||||
|
|
||||||
|
Runtime catalog entries include built-ins and configured `narratio.artifact.<name>` sources.
|
||||||
|
|
||||||
|
Catalog states:
|
||||||
|
- `planned`: source is registered and known for this run
|
||||||
|
- `executable`: configured artifact is selected for analyze execution
|
||||||
|
- `available`: artifact has a usable file path (generated this run or reused from disk)
|
||||||
|
|
||||||
|
Resolution behavior:
|
||||||
|
- built-in sources resolve via manifest producer outputs first, then canonical fallback path
|
||||||
|
- configured `narratio.artifact.<name>` sources resolve through runtime catalog availability
|
||||||
|
- configured source lookup requires catalog context
|
||||||
|
|
||||||
|
Configured artifact provenance values:
|
||||||
|
- `generated.current_analyze_run`
|
||||||
|
- `filesystem.disabled_artifact_output`
|
||||||
|
|
||||||
|
Content validation:
|
||||||
|
- transcript built-ins: JSON with top-level `segments` array
|
||||||
|
- bounds built-in: valid JSON
|
||||||
|
- configured artifacts: non-empty text file
|
||||||
|
|
||||||
|
## Skip and resume behavior
|
||||||
|
- resolver and catalog have no direct skip/resume decisions
|
||||||
|
- stage/runner skip-resume behavior consumes catalog/resolver results
|
||||||
|
|
||||||
|
## Failure behavior
|
||||||
|
- unsupported source -> source validation error
|
||||||
|
- known source unavailable -> `ErrSessionArtifactNotFound`
|
||||||
|
- configured source without catalog -> resolution error
|
||||||
|
- resolved file with invalid content -> validation error
|
||||||
|
|
||||||
|
## Tests to inspect before changing
|
||||||
|
- `internal/artifacts/artifact_resolver_test.go`
|
||||||
|
- `internal/artifacts/catalog_test.go`
|
||||||
|
- `internal/stage/analyze_test.go`
|
||||||
|
- `internal/config/scriptorium_test.go`
|
||||||
|
|
||||||
|
## Architectural invariants
|
||||||
|
- built-in IDs are static and registry-backed
|
||||||
|
- configured artifact IDs are runtime-derived (`narratio.artifact.<name>`) and catalog-backed
|
||||||
|
- built-in/source resolution remains deterministic and validation-gated
|
||||||
81
docs/internal/manifest.md
Normal file
81
docs/internal/manifest.md
Normal file
@@ -0,0 +1,81 @@
|
|||||||
|
# Internal: Manifest
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Describe Narratio's durable execution state model for session-level and run-level manifests, including lifecycle transitions and persistence behavior.
|
||||||
|
|
||||||
|
## Inputs and outputs
|
||||||
|
Inputs:
|
||||||
|
- Session identity and run identity from app orchestration.
|
||||||
|
- Stage transition events and stage result payloads.
|
||||||
|
|
||||||
|
Outputs:
|
||||||
|
- Session manifest at `{workspace.root}/work/{campaign}/{session_id}/manifest.json`.
|
||||||
|
- Run manifest at `{workspace.root}/work/{campaign}/{session_id}/runs/{run_id}/manifest.json`.
|
||||||
|
|
||||||
|
## Boundaries
|
||||||
|
Owns:
|
||||||
|
- Manifest schemas (`Manifest`, `RunManifest`, stage records, error records, input/artifact records).
|
||||||
|
- Stage status/action transition methods.
|
||||||
|
- Persistent store contract (`manifest.Store`) and local JSON store implementation.
|
||||||
|
|
||||||
|
Does not own:
|
||||||
|
- Stage implementation details.
|
||||||
|
- Path construction policy outside manifest file persistence calls.
|
||||||
|
- CLI command behavior.
|
||||||
|
|
||||||
|
## Config fields used
|
||||||
|
Manifest package itself does not read config directly.
|
||||||
|
|
||||||
|
Manifest identity fields are populated by app/stage orchestration from:
|
||||||
|
- `session.session_id`
|
||||||
|
- `session.campaign`
|
||||||
|
- `pipeline.workspace.root`
|
||||||
|
- `pipeline.storage.s3.*` (when archive/S3 identity is set)
|
||||||
|
|
||||||
|
## External adapters used
|
||||||
|
- No external service adapters.
|
||||||
|
- Uses local filesystem for persistence via `manifest.LocalStore`.
|
||||||
|
|
||||||
|
## State and manifest behavior
|
||||||
|
Session manifest model:
|
||||||
|
- Tracks durable per-session stage state and provenance (`pending`, `running`, `succeeded`, `failed`, `skipped`, `stale`, `interrupted`).
|
||||||
|
- Stores resolved inputs, durable artifacts, stage logs/config refs, and stage metadata.
|
||||||
|
|
||||||
|
Run manifest model:
|
||||||
|
- Tracks one invocation (`run_id`) with requested stages and force mode.
|
||||||
|
- Tracks per-stage action (`run` or `skip`) and per-stage status.
|
||||||
|
- Tracks overall run status (`running`, `succeeded`, `failed`).
|
||||||
|
|
||||||
|
Persistence behavior:
|
||||||
|
- Load validates required identity/timestamp fields and normalizes maps/records.
|
||||||
|
- Save updates `updated_at` and writes JSON atomically (temp file + rename).
|
||||||
|
- Session and run manifests are saved incrementally before/after stage transitions.
|
||||||
|
|
||||||
|
Relationship during execution:
|
||||||
|
- Runner updates both manifests for every stage transition.
|
||||||
|
- Session manifest is the durable pipeline-progress ledger.
|
||||||
|
- Run manifest is invocation history and audit record.
|
||||||
|
- Analyze stage outputs are persisted as `kind=scriptorium_artifact` with `source_id=narratio.artifact.<name>` for configured artifact identity.
|
||||||
|
|
||||||
|
## Skip and resume behavior
|
||||||
|
- Resume and skip decisions are based on session-manifest stage statuses.
|
||||||
|
- `--force` reruns selected stages and marks downstream succeeded stages as `stale` in session manifest.
|
||||||
|
- Run manifest records whether each stage was executed or skipped in that invocation.
|
||||||
|
|
||||||
|
## Failure behavior
|
||||||
|
- Stage failure marks both manifests failed for that stage and records error messages/timestamps.
|
||||||
|
- Save failures are returned immediately and fail the command.
|
||||||
|
- Invalid/malformed manifest files fail load with explicit validation/decode errors.
|
||||||
|
|
||||||
|
## Tests to inspect before changing
|
||||||
|
- `internal/manifest/manifest_test.go`
|
||||||
|
- `internal/manifest/run_manifest_test.go`
|
||||||
|
- `internal/manifest/store_test.go`
|
||||||
|
- `internal/app/runner_test.go`
|
||||||
|
- `internal/app/run_control_test.go`
|
||||||
|
- `internal/app/resume_run_stage_test.go`
|
||||||
|
|
||||||
|
## Architectural invariants
|
||||||
|
- Session manifest is authoritative for stage progression across invocations.
|
||||||
|
- Run manifest is invocation-scoped and never replaces session manifest as progress authority.
|
||||||
|
- Manifest writes are atomic and deterministic (JSON + newline, temp rename pattern).
|
||||||
84
docs/internal/stage-analyze.md
Normal file
84
docs/internal/stage-analyze.md
Normal file
@@ -0,0 +1,84 @@
|
|||||||
|
# Stage: analyze
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Execute selected configured Scriptorium artifacts in deterministic dependency order and promote successful outputs to canonical session artifact paths.
|
||||||
|
|
||||||
|
## Inputs and Outputs
|
||||||
|
Inputs:
|
||||||
|
- configured artifact definitions from `pipeline.scriptorium.artifacts`
|
||||||
|
- selected artifact filter from runtime (`--artifacts`) when provided
|
||||||
|
- resolved artifact input sources declared per artifact (`inputs.*.source`)
|
||||||
|
- optional previous-session file inputs (`previous_session_artifact`)
|
||||||
|
|
||||||
|
Outputs:
|
||||||
|
- one promoted output file per executed configured artifact at that artifact's configured `output_path`
|
||||||
|
- stage metadata containing generated artifact entries and reused disabled-artifact entries
|
||||||
|
|
||||||
|
## Boundaries
|
||||||
|
Owns:
|
||||||
|
- runtime artifact catalog construction for analyze execution
|
||||||
|
- selected-artifact planning and dependency ordering
|
||||||
|
- per-artifact input resolution, var resolution, timeout/render-debug resolution
|
||||||
|
- Scriptorium run/render invocation for each selected artifact
|
||||||
|
- run-local output generation and canonical promotion
|
||||||
|
|
||||||
|
Does not own:
|
||||||
|
- transcript generation/processing stages
|
||||||
|
- archive promotion policy
|
||||||
|
- per-artifact resume semantics
|
||||||
|
|
||||||
|
## Config Fields Used
|
||||||
|
- `session.session_id`
|
||||||
|
- `session.campaign`
|
||||||
|
- `pipeline.workspace.root`
|
||||||
|
- `pipeline.scriptorium.binary`
|
||||||
|
- `pipeline.scriptorium.config_path`
|
||||||
|
- `pipeline.scriptorium.timeout`
|
||||||
|
- `pipeline.scriptorium.render_debug`
|
||||||
|
- `pipeline.scriptorium.artifacts.<name>.*`
|
||||||
|
- `enabled`
|
||||||
|
- `depends_on`
|
||||||
|
- `prompt_id`
|
||||||
|
- `profile_id`
|
||||||
|
- `timeout`
|
||||||
|
- `output_path`
|
||||||
|
- `render_debug`
|
||||||
|
- `inputs`
|
||||||
|
- `vars`
|
||||||
|
|
||||||
|
## External Adapters Used
|
||||||
|
- Scriptorium adapter:
|
||||||
|
- optional `RenderArtifact` (render debug)
|
||||||
|
- `RunArtifact` (artifact generation)
|
||||||
|
|
||||||
|
## State and Manifest Behavior
|
||||||
|
- If `pipeline.scriptorium` is absent, stage returns success metadata with `skipped=true`.
|
||||||
|
- If no artifacts are configured, stage returns success metadata with `skipped=true`.
|
||||||
|
- If zero artifacts are executable after `enabled` + `--artifacts` filtering, stage returns success metadata with `skipped=true`.
|
||||||
|
- Builds runtime catalog with built-ins and configured artifacts.
|
||||||
|
- Non-executable configured artifacts are marked available only when their configured output file exists and is valid on disk.
|
||||||
|
- Executes selected configured artifacts in topological order with deterministic tie-breaking.
|
||||||
|
- For each generated artifact, records metadata fields including `name`, `source_id`, `output_kind`, `path`, `prompt_id`, `profile_id`, and `provenance`.
|
||||||
|
- Reused disabled artifacts are recorded separately in `reused_artifacts` with provenance `filesystem.disabled_artifact_output`.
|
||||||
|
|
||||||
|
## Skip and Resume Behavior
|
||||||
|
- Runner-level skip applies when analyze is already `succeeded` and `--force` is not set.
|
||||||
|
- Analyze remains stage-scoped for resume/skip; there is no per-artifact resume state.
|
||||||
|
- `--artifacts` filters which configured artifacts are executable when analyze runs; it does not imply `--force`.
|
||||||
|
|
||||||
|
## Failure Behavior
|
||||||
|
- Fails on invalid dependency ordering, unavailable required configured inputs, invalid built-in input prerequisites, render/run adapter failures, validation-failed adapter results, or missing/empty outputs.
|
||||||
|
- Required configured dependency missing from catalog availability fails clearly before invocation.
|
||||||
|
- Optional missing inputs are omitted.
|
||||||
|
|
||||||
|
## Tests to Inspect Before Changing
|
||||||
|
- `internal/stage/analyze_test.go`
|
||||||
|
- `internal/artifacts/catalog_test.go`
|
||||||
|
- `internal/artifacts/artifact_resolver_test.go`
|
||||||
|
- `internal/adapters/scriptorium/subprocess_test.go`
|
||||||
|
|
||||||
|
## Architectural Invariants
|
||||||
|
- Configured artifacts are identified by `narratio.artifact.<name>` source IDs.
|
||||||
|
- Artifact-to-artifact references rely on explicit `depends_on` declarations validated in config.
|
||||||
|
- Generated analyze outputs are treated uniformly as Scriptorium artifacts.
|
||||||
|
- Successful outputs must exist and be non-empty before promotion.
|
||||||
68
docs/internal/stage-archive.md
Normal file
68
docs/internal/stage-archive.md
Normal file
@@ -0,0 +1,68 @@
|
|||||||
|
# Stage: archive
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Publish run records and promoted session artifacts to object storage, then atomically advance the remote current pointer.
|
||||||
|
|
||||||
|
## Inputs and Outputs
|
||||||
|
Inputs:
|
||||||
|
- session manifest and prerequisite stage records
|
||||||
|
- run root contents under `runs/{run_id}/`
|
||||||
|
- promotion sources from session root (`archive.promote_artifacts`)
|
||||||
|
|
||||||
|
Outputs:
|
||||||
|
- uploaded run files under `{session_prefix}/runs/{run_id}/...`
|
||||||
|
- uploaded promoted artifacts under `{session_prefix}/...`
|
||||||
|
- `{session_prefix}/current/manifest.json`
|
||||||
|
- `{session_prefix}/current/run_id.txt` written last
|
||||||
|
|
||||||
|
## Boundaries
|
||||||
|
Owns:
|
||||||
|
- Archive enable/disable gate behavior
|
||||||
|
- Prerequisite stage success enforcement
|
||||||
|
- Run file collection and upload (excluding `audio/`)
|
||||||
|
- Promotion rule resolution and upload
|
||||||
|
- Commit pointer publish order
|
||||||
|
|
||||||
|
Does not own:
|
||||||
|
- Stage execution before archive
|
||||||
|
- Post-archive local cleanup policy execution (handled by app cleanup logic)
|
||||||
|
|
||||||
|
## Config Fields Used
|
||||||
|
- `pipeline.archive.enabled`
|
||||||
|
- `pipeline.archive.upload_run`
|
||||||
|
- `pipeline.archive.promote_artifacts`
|
||||||
|
- `pipeline.storage.s3.bucket`
|
||||||
|
- `pipeline.storage.s3.root_prefix`
|
||||||
|
- `pipeline.workspace.root`
|
||||||
|
- `session.campaign`
|
||||||
|
- `session.session_id`
|
||||||
|
|
||||||
|
## External Adapters Used
|
||||||
|
- Object storage backend (`env.ObjectStore`) for upload/list primitives.
|
||||||
|
|
||||||
|
## State and Manifest Behavior
|
||||||
|
- Requires `prepare`, `transcribe`, `merge`, `polish`, `normalize`, `trim`, and `analyze` status `succeeded`.
|
||||||
|
- Resolves bucket/prefix from manifest identity first, then config fallback.
|
||||||
|
- Writes metadata including:
|
||||||
|
- upload counts/paths
|
||||||
|
- `current_manifest_key`
|
||||||
|
- `current_run_id_key`
|
||||||
|
- `current_pointer_written`
|
||||||
|
- On skipped archive path, returns metadata with `skipped=true` and pointer not written.
|
||||||
|
|
||||||
|
## Skip and Resume Behavior
|
||||||
|
- Stage may self-skip (metadata skip) when archive disabled or run upload disabled.
|
||||||
|
- Runner-level skip also applies for previously succeeded stage unless forced.
|
||||||
|
|
||||||
|
## Failure Behavior
|
||||||
|
- Fails on missing prerequisite success, missing object store when required, missing run root, missing required promotion source, upload failures, or pointer write failures.
|
||||||
|
- Pointer semantics are fail-safe: `current/run_id.txt` is not written if prior required uploads fail.
|
||||||
|
|
||||||
|
## Tests to Inspect Before Changing
|
||||||
|
- `internal/stage/archive_test.go`
|
||||||
|
- `internal/app/post_archive_cleanup_test.go`
|
||||||
|
|
||||||
|
## Architectural Invariants
|
||||||
|
- Run upload excludes `audio/` subtree.
|
||||||
|
- `current/manifest.json` uploads before `current/run_id.txt`.
|
||||||
|
- `current/run_id.txt` is the remote publish commit marker.
|
||||||
63
docs/internal/stage-merge.md
Normal file
63
docs/internal/stage-merge.md
Normal file
@@ -0,0 +1,63 @@
|
|||||||
|
# Stage: merge
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Normalize per-speaker raw transcripts and merge them into one merged transcript via Seriatim.
|
||||||
|
|
||||||
|
## Inputs and Outputs
|
||||||
|
Inputs:
|
||||||
|
- `transcripts/raw/*.json`
|
||||||
|
- `inputs/speakers.yml`
|
||||||
|
- `inputs/autocorrect.yml`
|
||||||
|
|
||||||
|
Outputs:
|
||||||
|
- `transcripts/merged.json`
|
||||||
|
- optional `artifacts/seriatim.report.json` (when report enabled)
|
||||||
|
|
||||||
|
## Boundaries
|
||||||
|
Owns:
|
||||||
|
- Raw transcript discovery/validation
|
||||||
|
- Per-input normalize calls to Seriatim
|
||||||
|
- Final merge call to Seriatim
|
||||||
|
- Run-local log/config/report path wiring
|
||||||
|
- Promotion of merged/report outputs to canonical paths
|
||||||
|
|
||||||
|
Does not own:
|
||||||
|
- Transcript polishing or downstream artifact generation
|
||||||
|
|
||||||
|
## Config Fields Used
|
||||||
|
- `session.session_id`
|
||||||
|
- `session.campaign`
|
||||||
|
- `pipeline.workspace.root`
|
||||||
|
- `pipeline.seriatim.binary`
|
||||||
|
- `pipeline.seriatim.timeout`
|
||||||
|
- `pipeline.seriatim.output_schema`
|
||||||
|
- `pipeline.seriatim.coalesce_gap`
|
||||||
|
- `pipeline.seriatim.report`
|
||||||
|
- `pipeline.seriatim.env.*`
|
||||||
|
|
||||||
|
## External Adapters Used
|
||||||
|
- Seriatim adapter:
|
||||||
|
- `Normalize` for each raw input
|
||||||
|
- `Run` for final merge
|
||||||
|
|
||||||
|
## State and Manifest Behavior
|
||||||
|
- Reads transcript inputs from transcribe stage outputs in manifest when present; falls back to canonical raw directory.
|
||||||
|
- Writes run-local outputs/logs/config under `runs/{run_id}/merge/...` when enabled.
|
||||||
|
- Promotes canonical merged transcript and optional report.
|
||||||
|
- Records normalized-input provenance and adapter metadata in stage metadata.
|
||||||
|
|
||||||
|
## Skip and Resume Behavior
|
||||||
|
- Runner-level skip applies when already succeeded and not forced.
|
||||||
|
- Forced rerun of this or upstream stages can stale downstream succeeded stages via runner invalidation.
|
||||||
|
|
||||||
|
## Failure Behavior
|
||||||
|
- Fails on missing/invalid raw transcripts, missing speakers/autocorrect files, normalize failure, merge failure, invalid merged output JSON, or invalid report JSON when enabled.
|
||||||
|
|
||||||
|
## Tests to Inspect Before Changing
|
||||||
|
- `internal/stage/merge_test.go`
|
||||||
|
- `internal/adapters/seriatim/subprocess_test.go`
|
||||||
|
|
||||||
|
## Architectural Invariants
|
||||||
|
- Merge consumes normalized forms of each raw transcript.
|
||||||
|
- Merged transcript must validate before promotion.
|
||||||
|
- Report output is optional and gated by config.
|
||||||
56
docs/internal/stage-normalize.md
Normal file
56
docs/internal/stage-normalize.md
Normal file
@@ -0,0 +1,56 @@
|
|||||||
|
# Stage: normalize
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Normalize the processed transcript into a deterministic intermediate schema for trim and optionally emit a normalize report.
|
||||||
|
|
||||||
|
## Inputs and Outputs
|
||||||
|
Inputs:
|
||||||
|
- `transcripts/processed.json`
|
||||||
|
|
||||||
|
Outputs:
|
||||||
|
- `transcripts/normalized.json` (or configured normalize output path)
|
||||||
|
- optional `artifacts/seriatim.normalize.report.json`
|
||||||
|
|
||||||
|
## Boundaries
|
||||||
|
Owns:
|
||||||
|
- Processed transcript discovery/validation
|
||||||
|
- Normalize request construction and invocation
|
||||||
|
- Optional normalize report wiring
|
||||||
|
- Promotion of normalized transcript and optional report
|
||||||
|
|
||||||
|
Does not own:
|
||||||
|
- Bounds detection or segment trimming
|
||||||
|
|
||||||
|
## Config Fields Used
|
||||||
|
- `session.session_id`
|
||||||
|
- `session.campaign`
|
||||||
|
- `pipeline.workspace.root`
|
||||||
|
- `pipeline.normalize.output_path`
|
||||||
|
- `pipeline.normalize.output_schema`
|
||||||
|
- `pipeline.normalize.report`
|
||||||
|
- `pipeline.seriatim.binary`
|
||||||
|
- `pipeline.seriatim.timeout`
|
||||||
|
|
||||||
|
## External Adapters Used
|
||||||
|
- Seriatim adapter (`Normalize`).
|
||||||
|
|
||||||
|
## State and Manifest Behavior
|
||||||
|
- Reads processed transcript from polish outputs in manifest when present; falls back to canonical path.
|
||||||
|
- Uses run-local output/report/log/config paths when run layout is enabled.
|
||||||
|
- Promotes canonical normalized transcript and optional normalize report.
|
||||||
|
- Records adapter/result metadata including source path selection.
|
||||||
|
|
||||||
|
## Skip and Resume Behavior
|
||||||
|
- Runner-level skip applies when already succeeded and not forced.
|
||||||
|
- Forced reruns can stale downstream succeeded stages.
|
||||||
|
|
||||||
|
## Failure Behavior
|
||||||
|
- Fails on missing/invalid processed transcript, adapter error, invalid normalized output, or invalid report output when report enabled.
|
||||||
|
|
||||||
|
## Tests to Inspect Before Changing
|
||||||
|
- `internal/stage/normalize_test.go`
|
||||||
|
- `internal/adapters/seriatim/subprocess_test.go`
|
||||||
|
|
||||||
|
## Architectural Invariants
|
||||||
|
- Normalized output must validate as processed-transcript-compatible JSON (`segments` array required).
|
||||||
|
- Default normalize config is applied when `pipeline.normalize` is unset.
|
||||||
69
docs/internal/stage-polish.md
Normal file
69
docs/internal/stage-polish.md
Normal file
@@ -0,0 +1,69 @@
|
|||||||
|
# Stage: polish
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Polish merged transcript with Audita and produce a processed transcript for downstream normalization/analyze.
|
||||||
|
|
||||||
|
## Inputs and Outputs
|
||||||
|
Inputs:
|
||||||
|
- `transcripts/merged.json`
|
||||||
|
- `inputs/glossary.yml`
|
||||||
|
|
||||||
|
Outputs:
|
||||||
|
- `transcripts/processed.json`
|
||||||
|
- optional `artifacts/audita.report.json` (when report enabled)
|
||||||
|
|
||||||
|
## Boundaries
|
||||||
|
Owns:
|
||||||
|
- Merged transcript discovery/validation
|
||||||
|
- Audita invocation request construction
|
||||||
|
- Run-local logs/config/work-dir/report wiring
|
||||||
|
- Promotion of processed transcript and optional report
|
||||||
|
|
||||||
|
Does not own:
|
||||||
|
- Upstream merge normalization
|
||||||
|
- Downstream normalize/trim/analyze logic
|
||||||
|
|
||||||
|
## Config Fields Used
|
||||||
|
- `session.session_id`
|
||||||
|
- `session.campaign`
|
||||||
|
- `pipeline.workspace.root`
|
||||||
|
- `pipeline.audita.binary`
|
||||||
|
- `pipeline.audita.timeout`
|
||||||
|
- `pipeline.audita.llm_api_key_env`
|
||||||
|
- `pipeline.audita.modules`
|
||||||
|
- `pipeline.audita.base_url`
|
||||||
|
- `pipeline.audita.model`
|
||||||
|
- `pipeline.audita.transcript_description`
|
||||||
|
- `pipeline.audita.config_path`
|
||||||
|
- `pipeline.audita.output_schema`
|
||||||
|
- `pipeline.audita.work_dir_retention`
|
||||||
|
- `pipeline.audita.total_llm_concurrency`
|
||||||
|
- `pipeline.audita.proposal_llm_concurrency`
|
||||||
|
- `pipeline.audita.validation_model`
|
||||||
|
- `pipeline.audita.validation_llm_concurrency`
|
||||||
|
- `pipeline.audita.report`
|
||||||
|
|
||||||
|
## External Adapters Used
|
||||||
|
- Audita adapter (`env.Audita.Run`).
|
||||||
|
|
||||||
|
## State and Manifest Behavior
|
||||||
|
- Reads merged transcript from merge manifest outputs when available; falls back to canonical merged path.
|
||||||
|
- Uses run-local output/report/log/config/scratch paths when run layout is enabled.
|
||||||
|
- Promotes canonical `transcripts/processed.json` and optional report.
|
||||||
|
- Records adapter invocation metadata, credential presence signal, and output provenance in stage metadata.
|
||||||
|
|
||||||
|
## Skip and Resume Behavior
|
||||||
|
- Runner-level skip applies when already succeeded and not forced.
|
||||||
|
- Forced rerun can stale downstream succeeded stages via runner invalidation.
|
||||||
|
|
||||||
|
## Failure Behavior
|
||||||
|
- Fails on missing/invalid merged transcript, missing glossary, adapter error, invalid processed output shape (`segments` array required), or invalid report JSON when enabled.
|
||||||
|
|
||||||
|
## Tests to Inspect Before Changing
|
||||||
|
- `internal/stage/polish_test.go`
|
||||||
|
- `internal/adapters/audita/subprocess_test.go`
|
||||||
|
|
||||||
|
## Architectural Invariants
|
||||||
|
- Processed transcript must contain a top-level `segments` array.
|
||||||
|
- Report behavior is strictly config-gated.
|
||||||
|
- Stage output canonicalization always ends at `transcripts/processed.json`.
|
||||||
74
docs/internal/stage-prepare.md
Normal file
74
docs/internal/stage-prepare.md
Normal file
@@ -0,0 +1,74 @@
|
|||||||
|
# Stage: prepare
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Materialize all required session inputs into canonical local workspace paths and record input provenance in the session manifest.
|
||||||
|
|
||||||
|
## Inputs and Outputs
|
||||||
|
Inputs:
|
||||||
|
- `session.yml` (resolved session config)
|
||||||
|
- `pipeline.resolved.yml` (materialized from resolved pipeline config)
|
||||||
|
- `speakers.yml`
|
||||||
|
- `autocorrect.yml`
|
||||||
|
- `glossary.yml`
|
||||||
|
- audio source:
|
||||||
|
- local (`session.inputs.audio_dir` or `session.inputs.audio_files`), or
|
||||||
|
- S3 (`session.inputs.audio_s3.prefix`)
|
||||||
|
|
||||||
|
Outputs:
|
||||||
|
- `inputs/session.yml`
|
||||||
|
- `inputs/pipeline.resolved.yml`
|
||||||
|
- `inputs/speakers.yml`
|
||||||
|
- `inputs/autocorrect.yml`
|
||||||
|
- `inputs/glossary.yml`
|
||||||
|
- `audio/*.flac` in session workdir
|
||||||
|
- `manifest.Inputs` records with checksums and source metadata
|
||||||
|
|
||||||
|
## Boundaries
|
||||||
|
Owns:
|
||||||
|
- Input path resolution and validation
|
||||||
|
- Local copy/materialization of configs and audio files
|
||||||
|
- S3 audio download to run-scoped spool, then copy into work audio dir
|
||||||
|
|
||||||
|
Does not own:
|
||||||
|
- Transcript generation/processing
|
||||||
|
- Archive publish behavior
|
||||||
|
|
||||||
|
## Config Fields Used
|
||||||
|
- `session.session_id`
|
||||||
|
- `session.campaign`
|
||||||
|
- `session.inputs.speakers_file`
|
||||||
|
- `session.inputs.autocorrect_file`
|
||||||
|
- `session.inputs.glossary_file`
|
||||||
|
- `session.inputs.audio_dir`
|
||||||
|
- `session.inputs.audio_files`
|
||||||
|
- `session.inputs.audio_s3.prefix`
|
||||||
|
- `pipeline.workspace.root`
|
||||||
|
- `pipeline.spool.root`
|
||||||
|
- `pipeline.storage.s3.bucket`
|
||||||
|
- `pipeline.storage.s3.root_prefix`
|
||||||
|
|
||||||
|
## External Adapters Used
|
||||||
|
- Object storage backend (`env.ObjectStore`) for S3 audio list/download when `audio_s3` is configured.
|
||||||
|
|
||||||
|
## State and Manifest Behavior
|
||||||
|
- Ensures workspace layout exists.
|
||||||
|
- Writes resolved config and input files to canonical `inputs/` paths.
|
||||||
|
- Records all prepared inputs into `manifest.Inputs` (sorted deterministically by kind/path).
|
||||||
|
- For S3 audio, records `S3Bucket`, `S3Key`, `S3Size`, `S3ETag`, and `SpoolPath` in each audio input record.
|
||||||
|
|
||||||
|
## Skip and Resume Behavior
|
||||||
|
- Runner-level skip applies when stage already `succeeded` and `--force` is not set.
|
||||||
|
- Stage itself is deterministic/idempotent for unchanged inputs (`copyFileIfChanged`, `writeBytesIfChanged`).
|
||||||
|
|
||||||
|
## Failure Behavior
|
||||||
|
- Fails on missing required files, invalid audio source combinations, no discoverable `.flac` files, duplicate audio basenames, missing object store for S3 mode, or S3 list/download failures.
|
||||||
|
|
||||||
|
## Tests to Inspect Before Changing
|
||||||
|
- `internal/stage/prepare_test.go`
|
||||||
|
- `internal/app/session_cli_test.go`
|
||||||
|
- `internal/config/load_validate_test.go`
|
||||||
|
|
||||||
|
## Architectural Invariants
|
||||||
|
- `audio_dir`/`audio_files` and `audio_s3` are mutually exclusive.
|
||||||
|
- Audio files must be `.flac`.
|
||||||
|
- Canonical `inputs/*` and `audio/*` paths are the durable source for downstream stages.
|
||||||
58
docs/internal/stage-transcribe.md
Normal file
58
docs/internal/stage-transcribe.md
Normal file
@@ -0,0 +1,58 @@
|
|||||||
|
# Stage: transcribe
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Generate per-speaker raw transcripts from prepared audio using WhisperX.
|
||||||
|
|
||||||
|
## Inputs and Outputs
|
||||||
|
Inputs:
|
||||||
|
- `audio/*.flac` prepared by `prepare`
|
||||||
|
|
||||||
|
Outputs:
|
||||||
|
- `transcripts/raw/<speaker>.json` for each input audio file
|
||||||
|
|
||||||
|
## Boundaries
|
||||||
|
Owns:
|
||||||
|
- Discovering prepared audio inputs
|
||||||
|
- Deriving speaker ids from audio basenames
|
||||||
|
- Parallel WhisperX invocation with bounded concurrency
|
||||||
|
- Validating produced JSON and promoting run-local outputs
|
||||||
|
|
||||||
|
Does not own:
|
||||||
|
- Transcript merge/polish/normalize/trim/analyze
|
||||||
|
|
||||||
|
## Config Fields Used
|
||||||
|
- `session.session_id`
|
||||||
|
- `session.campaign`
|
||||||
|
- `pipeline.workspace.root`
|
||||||
|
- `pipeline.whisperx.transcribe_url`
|
||||||
|
- `pipeline.whisperx.language`
|
||||||
|
- `pipeline.whisperx.timeout`
|
||||||
|
- `pipeline.whisperx.retries`
|
||||||
|
- `pipeline.whisperx.retry_delay`
|
||||||
|
- `pipeline.whisperx.concurrency`
|
||||||
|
|
||||||
|
## External Adapters Used
|
||||||
|
- WhisperX adapter (`env.WhisperX.Transcribe`).
|
||||||
|
|
||||||
|
## State and Manifest Behavior
|
||||||
|
- Uses run-local output paths under `runs/{run_id}/transcribe/outputs/...` when run layout is enabled.
|
||||||
|
- Validates each generated transcript JSON before promotion.
|
||||||
|
- Promotes canonical outputs to `transcripts/raw/*.json`.
|
||||||
|
- Records per-file metadata (attempts/status/duration/output path) in stage metadata.
|
||||||
|
|
||||||
|
## Skip and Resume Behavior
|
||||||
|
- Runner-level skip applies for previously succeeded stage unless forced.
|
||||||
|
- On forced upstream reruns, downstream succeeded stages can be marked `stale` by runner logic.
|
||||||
|
|
||||||
|
## Failure Behavior
|
||||||
|
- Fails if no prepared audio exists, duplicate speaker basenames are detected, adapter output path mismatches expected path, any output JSON is invalid, or one worker fails.
|
||||||
|
- Cancels in-flight workers after first terminal error.
|
||||||
|
|
||||||
|
## Tests to Inspect Before Changing
|
||||||
|
- `internal/stage/transcribe_test.go`
|
||||||
|
- `internal/app/whisperx_wiring_test.go`
|
||||||
|
|
||||||
|
## Architectural Invariants
|
||||||
|
- Speaker identity is derived from `.flac` basename and must be unique.
|
||||||
|
- Every successful speaker output must be valid JSON before promotion.
|
||||||
|
- Canonical raw transcript set is the only supported merge input surface.
|
||||||
75
docs/internal/stage-trim.md
Normal file
75
docs/internal/stage-trim.md
Normal file
@@ -0,0 +1,75 @@
|
|||||||
|
# Stage: trim
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Optionally trim the normalized transcript to session bounds; always produce a durable trimmed transcript.
|
||||||
|
|
||||||
|
## Inputs and Outputs
|
||||||
|
Inputs:
|
||||||
|
- `transcripts/normalized.json`
|
||||||
|
|
||||||
|
Outputs:
|
||||||
|
- `transcripts/trimmed.json` (or configured trim output path)
|
||||||
|
- when trim enabled: `artifacts/session_bounds.json`
|
||||||
|
|
||||||
|
## Boundaries
|
||||||
|
Owns:
|
||||||
|
- Trim-enabled switch behavior
|
||||||
|
- Bounds generation via Scriptorium artifact run
|
||||||
|
- Bounds validation against normalized transcript
|
||||||
|
- Keep-selector derivation and Seriatim trim invocation
|
||||||
|
- Copy-through behavior when disabled or bounds indicate unchanged transcript
|
||||||
|
|
||||||
|
Does not own:
|
||||||
|
- Upstream normalization
|
||||||
|
- Downstream artifact analysis
|
||||||
|
|
||||||
|
## Config Fields Used
|
||||||
|
- `session.session_id`
|
||||||
|
- `session.campaign`
|
||||||
|
- `pipeline.workspace.root`
|
||||||
|
- `pipeline.trim.enabled`
|
||||||
|
- `pipeline.trim.output_path`
|
||||||
|
- `pipeline.trim.bounds.prompt_id`
|
||||||
|
- `pipeline.trim.bounds.profile_id`
|
||||||
|
- `pipeline.trim.bounds.timeout`
|
||||||
|
- `pipeline.trim.bounds.output_path`
|
||||||
|
- `pipeline.trim.bounds.transcript_input_name`
|
||||||
|
- `pipeline.trim.bounds.render_debug`
|
||||||
|
- `pipeline.trim.bounds.render_output_path`
|
||||||
|
- `pipeline.seriatim.binary`
|
||||||
|
- `pipeline.seriatim.timeout`
|
||||||
|
- `pipeline.scriptorium.binary`
|
||||||
|
- `pipeline.scriptorium.config_path`
|
||||||
|
- `pipeline.scriptorium.timeout`
|
||||||
|
|
||||||
|
## External Adapters Used
|
||||||
|
- Scriptorium adapter:
|
||||||
|
- optional `RenderArtifact` for bounds debug render
|
||||||
|
- `RunArtifact` for bounds output
|
||||||
|
- Seriatim adapter:
|
||||||
|
- `Trim` when bounds indicate trimming is required
|
||||||
|
|
||||||
|
## State and Manifest Behavior
|
||||||
|
- Reads normalized transcript from normalize manifest outputs when available; falls back to canonical path.
|
||||||
|
- Uses run-local outputs/logs/reports/config/scratch paths when run layout is enabled.
|
||||||
|
- Promotes canonical trimmed transcript; promotes session bounds when trim enabled.
|
||||||
|
- Records bounds diagnostics, trim action, keep selector, and adapter metadata.
|
||||||
|
|
||||||
|
## Skip and Resume Behavior
|
||||||
|
- Runner-level skip applies when already succeeded and not forced.
|
||||||
|
- Forced reruns can stale downstream succeeded stages.
|
||||||
|
- When `trim.enabled=false`, stage still succeeds by copying normalized to trimmed output.
|
||||||
|
|
||||||
|
## Failure Behavior
|
||||||
|
- Fails on missing/invalid normalized transcript.
|
||||||
|
- With trim enabled, fails on missing adapters/config, bounds generation/validation errors, invalid bounds JSON, invalid range/segment ids, trim adapter failures, or invalid trimmed output.
|
||||||
|
|
||||||
|
## Tests to Inspect Before Changing
|
||||||
|
- `internal/stage/trim_test.go`
|
||||||
|
- `internal/adapters/scriptorium/subprocess_test.go`
|
||||||
|
- `internal/adapters/seriatim/subprocess_test.go`
|
||||||
|
|
||||||
|
## Architectural Invariants
|
||||||
|
- Trim never falls back to processed transcript; normalized transcript is required input.
|
||||||
|
- `session_bounds` output exists only for enabled trim path.
|
||||||
|
- Render-debug artifacts are diagnostics and not declared stage outputs.
|
||||||
71
docs/internal/storage.md
Normal file
71
docs/internal/storage.md
Normal file
@@ -0,0 +1,71 @@
|
|||||||
|
# Internal: Storage
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Document Narratio's remote storage backend contracts and implementations under `internal/adapters/storage`.
|
||||||
|
|
||||||
|
## Inputs and outputs
|
||||||
|
Inputs:
|
||||||
|
- Resolved storage config (`pipeline.storage.*`).
|
||||||
|
- Bucket-relative object keys and local file paths from stage/app orchestration.
|
||||||
|
|
||||||
|
Outputs:
|
||||||
|
- Listed/downloaded/uploaded object metadata (`ObjectInfo`).
|
||||||
|
- Existence checks and storage-layer errors.
|
||||||
|
|
||||||
|
## Boundaries
|
||||||
|
Owns:
|
||||||
|
- Remote object-store interface and implementation details.
|
||||||
|
- S3 client wiring and API calls.
|
||||||
|
- Object key normalization and upload/download/list primitives.
|
||||||
|
|
||||||
|
Does not own:
|
||||||
|
- Session/run prefix semantics.
|
||||||
|
- Archive commit order semantics.
|
||||||
|
- Manifest updates.
|
||||||
|
|
||||||
|
## Config fields used
|
||||||
|
- `pipeline.storage.backend`
|
||||||
|
- `pipeline.storage.s3.bucket`
|
||||||
|
- `pipeline.storage.s3.region`
|
||||||
|
- `pipeline.storage.s3.endpoint`
|
||||||
|
- `pipeline.storage.s3.force_path_style`
|
||||||
|
- `pipeline.storage.s3.access_key_id_env`
|
||||||
|
- `pipeline.storage.s3.secret_access_key_env`
|
||||||
|
|
||||||
|
## External adapters used
|
||||||
|
Storage package contracts:
|
||||||
|
- `ObjectStore` (active remote object-store boundary): `List`, `Download`, `Upload`, `Exists`.
|
||||||
|
- `Backend` (archive request boundary): currently implemented with `NoopBackend` only.
|
||||||
|
|
||||||
|
Implementations:
|
||||||
|
- `S3Backend`: AWS SDK-backed `ObjectStore` implementation.
|
||||||
|
- `FakeBackend`: deterministic test `ObjectStore` and archive backend.
|
||||||
|
- `NoopBackend`: deterministic no-op archive backend for compatibility wiring.
|
||||||
|
|
||||||
|
## State and manifest behavior
|
||||||
|
- Storage implementations are stateless with respect to manifest/session lifecycle.
|
||||||
|
- Caller supplies fully-qualified bucket-relative keys.
|
||||||
|
- Storage layer does not infer campaign/session/run/root-prefix semantics.
|
||||||
|
- Caller controls publish ordering; storage layer executes individual operations in the order invoked.
|
||||||
|
|
||||||
|
## Skip and resume behavior
|
||||||
|
- No storage-level skip/resume behavior.
|
||||||
|
- Skip/resume decisions are made by stage/app logic before storage calls occur.
|
||||||
|
|
||||||
|
## Failure behavior
|
||||||
|
- `NewObjectStoreFromConfig` fails when no remote backend is configured or required S3 config is missing.
|
||||||
|
- `S3Backend` constructor fails when required bucket is missing or AWS client setup fails.
|
||||||
|
- CRUD operations return contextual errors (including not-found behavior via `Exists`).
|
||||||
|
- Key normalization is applied before operations (`\\` to `/`, leading slash trimmed).
|
||||||
|
|
||||||
|
## Tests to inspect before changing
|
||||||
|
- `internal/adapters/storage/factory_test.go`
|
||||||
|
- `internal/adapters/storage/s3_backend_test.go`
|
||||||
|
- `internal/adapters/storage/fake_test.go`
|
||||||
|
- `internal/adapters/storage/keys_test.go`
|
||||||
|
- `internal/adapters/storage/archive.go` + consumers in stage tests (`prepare`, `archive`)
|
||||||
|
|
||||||
|
## Architectural invariants
|
||||||
|
- Callers pass full bucket-relative keys.
|
||||||
|
- Storage backends must not prepend or infer narratio prefixes.
|
||||||
|
- Remote transport details remain isolated to storage adapter implementations.
|
||||||
68
docs/internal/workspace.md
Normal file
68
docs/internal/workspace.md
Normal file
@@ -0,0 +1,68 @@
|
|||||||
|
# Workspace internals
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Define the local durable and run-local workspace model used by stages, manifests, resume, and archive.
|
||||||
|
|
||||||
|
## Inputs and Outputs
|
||||||
|
Inputs:
|
||||||
|
- `pipeline.workspace.root`
|
||||||
|
- `session.campaign`
|
||||||
|
- `session.session_id`
|
||||||
|
- generated `run_id`
|
||||||
|
|
||||||
|
Outputs:
|
||||||
|
- Session manifest at `{workspace.root}/work/{campaign}/{session_id}/manifest.json`
|
||||||
|
- Run manifest at `{workspace.root}/work/{campaign}/{session_id}/runs/{run_id}/manifest.json`
|
||||||
|
- Canonical durable session directories and run-local stage trees
|
||||||
|
|
||||||
|
## Boundaries
|
||||||
|
Owns:
|
||||||
|
- Session-level path layout (`inputs/`, `audio/`, `transcripts/`, `artifacts/`, `reports/`, `logs/`, `config/`, `current/`, `runs/`)
|
||||||
|
- Run-local stage sandbox layout under `runs/{run_id}/{stage}/`
|
||||||
|
- Session lock acquisition/release (`.lock`)
|
||||||
|
|
||||||
|
Does not own:
|
||||||
|
- Stage business logic
|
||||||
|
- Remote archive semantics (documented in `stage-archive.md`)
|
||||||
|
- CLI argument parsing
|
||||||
|
|
||||||
|
## Config Fields Used
|
||||||
|
- `pipeline.workspace.root`
|
||||||
|
- `pipeline.workspace.cleanup_after_archive`
|
||||||
|
- `pipeline.spool.root`
|
||||||
|
- `pipeline.spool.delete_audio_after_archive`
|
||||||
|
- `session.campaign`
|
||||||
|
- `session.session_id`
|
||||||
|
|
||||||
|
## External Adapters Used
|
||||||
|
None directly in this subsystem. Stages may use object storage adapters and then write local outputs into this layout.
|
||||||
|
|
||||||
|
## State and Manifest Behavior
|
||||||
|
- Session state is persisted in the session manifest (`manifest.Manifest`).
|
||||||
|
- Invocation history is persisted per run in run manifests under `runs/{run_id}/manifest.json`.
|
||||||
|
- During each run, stage outputs are often written run-local first (`runs/{run_id}/{stage}/outputs/...`) and promoted to canonical session paths after stage success.
|
||||||
|
- `manifest.Artifacts` entries record `ProducerRunID` for durable outputs.
|
||||||
|
- For S3 audio sessions, `prepare` records spool/work paths and S3 provenance in `manifest.Inputs`.
|
||||||
|
|
||||||
|
## Skip and Resume Behavior
|
||||||
|
- Skip/resume decisions are made in `internal/app` (`run_control.go`, `resume.go`) using stage status in the session manifest.
|
||||||
|
- `--force` reruns selected stages and marks downstream previously-succeeded stages as `stale`.
|
||||||
|
- Workspace layout is idempotent (`EnsureLayoutFor`) and reused across runs.
|
||||||
|
|
||||||
|
## Failure Behavior
|
||||||
|
- Failures preserve manifests and run-local files for inspection.
|
||||||
|
- Lock conflicts fail fast via `ErrLockConflict`.
|
||||||
|
- Cleanup can fail post-archive; failure is recorded in archive stage metadata and returned by the run.
|
||||||
|
|
||||||
|
## Tests to Inspect Before Changing
|
||||||
|
- `internal/artifacts/local_test.go`
|
||||||
|
- `internal/stage/run_local_test.go`
|
||||||
|
- `internal/app/run_control_test.go`
|
||||||
|
- `internal/app/resume_run_stage_test.go`
|
||||||
|
- `internal/app/post_archive_cleanup_test.go`
|
||||||
|
|
||||||
|
## Architectural Invariants
|
||||||
|
- Session root is campaign-aware: `{workspace.root}/work/{campaign}/{session_id}`.
|
||||||
|
- Run roots are always nested: `runs/{run_id}` under the session root.
|
||||||
|
- Run-local output promotion must end in canonical session paths.
|
||||||
|
- Cleanup only targets run-scoped directories and must never delete configured root directories.
|
||||||
149
docs/operations.md
Normal file
149
docs/operations.md
Normal file
@@ -0,0 +1,149 @@
|
|||||||
|
# Operations
|
||||||
|
|
||||||
|
This guide describes the implemented operator lifecycle for Narratio.
|
||||||
|
|
||||||
|
For field-level configuration, see [docs/config.md](./config.md). For full command/flag reference, see [docs/cli.md](./cli.md).
|
||||||
|
|
||||||
|
## Normal workflow (S3-first path)
|
||||||
|
|
||||||
|
1. Upload session `.flac` files to object storage under the session audio prefix.
|
||||||
|
2. Run Narratio:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run --session-id 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
3. Read success output:
|
||||||
|
- `narratio run: session <session_id>; executed=<n> skipped=<n>; manifest=<path>`
|
||||||
|
- use `manifest=<path>` with `status` for inspection.
|
||||||
|
|
||||||
|
Notes:
|
||||||
|
- default config/session discovery applies unless `--config` and `--session` are passed.
|
||||||
|
- S3 audio mode requires `session.inputs.audio_s3.prefix` and valid object-store access.
|
||||||
|
|
||||||
|
## Local filesystem layout and state artifacts
|
||||||
|
|
||||||
|
Session root:
|
||||||
|
- `{workspace.root}/work/{campaign}/{session_id}/`
|
||||||
|
|
||||||
|
Primary state:
|
||||||
|
- `manifest.json`: session-level stage state.
|
||||||
|
- `runs/{run_id}/manifest.json`: invocation-level state.
|
||||||
|
- `.lock`: session lock while a run is active.
|
||||||
|
|
||||||
|
Canonical session directories:
|
||||||
|
- `inputs/`
|
||||||
|
- `audio/`
|
||||||
|
- `transcripts/`
|
||||||
|
- `artifacts/`
|
||||||
|
- `reports/`
|
||||||
|
- `logs/`
|
||||||
|
- `config/`
|
||||||
|
- `current/`
|
||||||
|
- `runs/`
|
||||||
|
|
||||||
|
Run-local stage directories:
|
||||||
|
- `runs/{run_id}/{stage}/` with stage-local `outputs/`, `logs/`, `reports/`, `config/`, `scratch/`.
|
||||||
|
|
||||||
|
Behavior:
|
||||||
|
- directory creation is idempotent.
|
||||||
|
- stage outputs are generally generated run-local first, then promoted to canonical paths on success.
|
||||||
|
|
||||||
|
## Analyze artifact execution lifecycle
|
||||||
|
|
||||||
|
Analyze executes configured artifacts from `pipeline.scriptorium.artifacts`.
|
||||||
|
|
||||||
|
Execution model:
|
||||||
|
- executable set = enabled artifacts, filtered by `--artifacts` when provided.
|
||||||
|
- artifact-to-artifact dependencies are declared via `depends_on`.
|
||||||
|
- selected artifacts run in deterministic dependency order.
|
||||||
|
- after each successful artifact run, output is promoted to configured canonical `output_path`.
|
||||||
|
|
||||||
|
Configured artifact source reuse:
|
||||||
|
- a non-executable configured artifact can satisfy inputs if its configured output file already exists and is valid.
|
||||||
|
- reused configured artifact provenance is `filesystem.disabled_artifact_output`.
|
||||||
|
|
||||||
|
`--artifacts` behavior:
|
||||||
|
- accepted on `run`, `resume`, and `run-stage analyze`.
|
||||||
|
- filters analyze execution only; does not force stage rerun.
|
||||||
|
|
||||||
|
## Remote archive layout and publish contract
|
||||||
|
|
||||||
|
When archive is enabled and run upload is enabled, archive publishes under:
|
||||||
|
|
||||||
|
- session prefix: `{root_prefix}/campaigns/{campaign}/sessions/{session_id}/`
|
||||||
|
- run prefix: `{session_prefix}/runs/{run_id}/`
|
||||||
|
|
||||||
|
Archive uploads:
|
||||||
|
- run record files from run root (excluding `audio/`).
|
||||||
|
- promoted files from explicit `archive.promote_artifacts` rules.
|
||||||
|
|
||||||
|
Publish order:
|
||||||
|
1. upload `current/manifest.json`
|
||||||
|
2. upload `current/run_id.txt` last
|
||||||
|
|
||||||
|
`current/run_id.txt` is the remote commit marker.
|
||||||
|
|
||||||
|
Archive promotion is explicit and path-based:
|
||||||
|
- Narratio does not auto-promote all generated analyze artifacts.
|
||||||
|
- missing required promotion sources fail archive stage.
|
||||||
|
- missing optional promotion sources are skipped.
|
||||||
|
|
||||||
|
## Resume, retry, and safe rerun behavior
|
||||||
|
|
||||||
|
Default skip:
|
||||||
|
- `run` and `run-stage` skip already-succeeded stages unless `--force` is set.
|
||||||
|
|
||||||
|
Resume:
|
||||||
|
- `resume` starts at first non-succeeded stage.
|
||||||
|
- `resume --force` runs full stage order.
|
||||||
|
|
||||||
|
Forced reruns:
|
||||||
|
- force-rerunning an upstream succeeded stage marks downstream succeeded stages as `stale`.
|
||||||
|
|
||||||
|
Safe rerun pattern:
|
||||||
|
1. rerun the changed stage with `--force`.
|
||||||
|
2. run `resume` to rebuild downstream stages.
|
||||||
|
|
||||||
|
## Cleanup behavior
|
||||||
|
|
||||||
|
Cleanup is considered only when archive stage executed and succeeded.
|
||||||
|
|
||||||
|
Cleanup toggles:
|
||||||
|
- `pipeline.spool.delete_audio_after_archive=true` deletes run-scoped spool audio.
|
||||||
|
- `pipeline.workspace.cleanup_after_archive=true` deletes run-scoped local run directory.
|
||||||
|
|
||||||
|
Cleanup eligibility gates:
|
||||||
|
- archive enabled
|
||||||
|
- archive run upload enabled
|
||||||
|
- run record upload completed
|
||||||
|
- current pointer write completed (`current/run_id.txt` written)
|
||||||
|
|
||||||
|
No cleanup for failed/incomplete/unarchived/archive-skipped runs.
|
||||||
|
|
||||||
|
## Failure and recovery playbooks
|
||||||
|
|
||||||
|
After failure, Narratio keeps:
|
||||||
|
- session manifest
|
||||||
|
- run manifest
|
||||||
|
- run-local artifacts/logs/config/reports
|
||||||
|
|
||||||
|
Failed or incomplete runs remain local-only.
|
||||||
|
|
||||||
|
Recommended recovery:
|
||||||
|
|
||||||
|
1. inspect state:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio status --manifest <manifest-path>
|
||||||
|
```
|
||||||
|
|
||||||
|
2. fix root cause (config/input/credentials/service availability).
|
||||||
|
3. continue with `resume`, or targeted `run-stage --force` followed by `resume`.
|
||||||
|
|
||||||
|
## Operational caveats
|
||||||
|
|
||||||
|
- `status` requires explicit `--manifest`; there is no session-id lookup command.
|
||||||
|
- local and S3 audio input modes are mutually exclusive.
|
||||||
|
- archive publish requires upstream stages through `analyze` to be `succeeded`.
|
||||||
|
- required promotion rules can fail when selected analyze artifacts did not generate a required file path.
|
||||||
762
docs/roadmap/runtime-artifacts.md
Normal file
762
docs/roadmap/runtime-artifacts.md
Normal file
@@ -0,0 +1,762 @@
|
|||||||
|
# Roadmap: Runtime-Defined Scriptorium Artifacts
|
||||||
|
|
||||||
|
## Status
|
||||||
|
|
||||||
|
Implementation roadmap for a pre-release hard cutover.
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
|
||||||
|
Narratio currently treats artifact generation as a narrow `analyze` stage that supports a hard-coded `session_recap` artifact. This roadmap describes how to generalize artifact generation so operators can define Scriptorium-backed output artifacts at runtime through `pipeline.yml`.
|
||||||
|
|
||||||
|
The goal is to keep Narratio as a fixed pipeline orchestrator while making the artifact generation step configurable, composable, deterministic, and easy to regenerate selectively.
|
||||||
|
|
||||||
|
## Desired Outcome
|
||||||
|
|
||||||
|
Operators should be able to define artifacts such as session recaps, player handouts, NPC summaries, quest logs, entity maps, or other campaign-specific outputs without changing Narratio code.
|
||||||
|
|
||||||
|
A configured artifact is declared under:
|
||||||
|
|
||||||
|
```text
|
||||||
|
pipeline.scriptorium.artifacts.<name>
|
||||||
|
```
|
||||||
|
|
||||||
|
Each configured artifact becomes a canonical runtime artifact source ID:
|
||||||
|
|
||||||
|
```text
|
||||||
|
narratio.artifact.<name>
|
||||||
|
```
|
||||||
|
|
||||||
|
For example:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
scriptorium:
|
||||||
|
artifacts:
|
||||||
|
session_recap:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd_session.session_recap
|
||||||
|
output_path: artifacts/session_recap.md
|
||||||
|
inputs:
|
||||||
|
transcript:
|
||||||
|
source: narratio.transcript.trimmed
|
||||||
|
required: true
|
||||||
|
```
|
||||||
|
|
||||||
|
This artifact is addressable by later artifacts as:
|
||||||
|
|
||||||
|
```text
|
||||||
|
narratio.artifact.session_recap
|
||||||
|
```
|
||||||
|
|
||||||
|
A dependent artifact can then consume it explicitly:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
scriptorium:
|
||||||
|
artifacts:
|
||||||
|
player_handout:
|
||||||
|
enabled: true
|
||||||
|
depends_on:
|
||||||
|
- session_recap
|
||||||
|
prompt_id: dnd_session.player_handout
|
||||||
|
output_path: artifacts/player_handout.md
|
||||||
|
inputs:
|
||||||
|
recap:
|
||||||
|
source: narratio.artifact.session_recap
|
||||||
|
required: true
|
||||||
|
transcript:
|
||||||
|
source: narratio.transcript.trimmed
|
||||||
|
required: true
|
||||||
|
```
|
||||||
|
|
||||||
|
## Resolved Design Decisions
|
||||||
|
|
||||||
|
The following decisions are settled for the initial implementation:
|
||||||
|
|
||||||
|
1. Configured artifact outputs must live under Narratio's internal artifact output directory, initially `artifacts/`.
|
||||||
|
2. The artifact output directory should be defined as an internal default in `internal/config/defaults.go`, but no public configuration knob should be exposed yet.
|
||||||
|
3. Artifact `output_path` should remain explicit in the initial implementation to avoid guessing file extensions or output formats.
|
||||||
|
4. A disabled artifact may still be referenced as an input if its declared output already exists on disk and passes basic validation.
|
||||||
|
5. A disabled artifact is not executable during the current analyze run.
|
||||||
|
6. Artifact-to-artifact references require an explicit `depends_on` entry. Narratio should fail fast if the dependency declaration is missing.
|
||||||
|
7. The manifest remains stage-oriented: `analyze` succeeds or fails as a full stage.
|
||||||
|
8. Analyze-stage metadata may record per-artifact output details for provenance and later resolution, but not for intra-stage resume semantics.
|
||||||
|
9. `--artifacts` should be added as a CLI filter for selective artifact generation.
|
||||||
|
10. `--artifacts` does not imply `--force`; it only changes which configured artifacts are treated as executable when `analyze` actually runs.
|
||||||
|
11. Because Narratio is still pre-release, the hard-coded `session_recap` behavior should be removed immediately rather than deprecated gradually.
|
||||||
|
|
||||||
|
## Scope
|
||||||
|
|
||||||
|
This roadmap covers:
|
||||||
|
|
||||||
|
- introducing a runtime artifact catalog;
|
||||||
|
- generalizing configured Scriptorium artifact execution;
|
||||||
|
- supporting `narratio.artifact.<name>` source IDs;
|
||||||
|
- adding explicit artifact dependencies;
|
||||||
|
- supporting disabled-but-resolvable artifact inputs;
|
||||||
|
- adding selective artifact execution via `--artifacts`;
|
||||||
|
- recording generated artifacts in analyze-stage metadata and/or manifest outputs;
|
||||||
|
- removing hard-coded `session_recap` behavior;
|
||||||
|
- updating tests and documentation.
|
||||||
|
|
||||||
|
## Non-Goals
|
||||||
|
|
||||||
|
This feature should not turn Narratio into a general workflow engine.
|
||||||
|
|
||||||
|
The initial implementation should not add:
|
||||||
|
|
||||||
|
- arbitrary shell-command artifacts;
|
||||||
|
- arbitrary user-defined stages;
|
||||||
|
- loops or conditional branching;
|
||||||
|
- automatic archive promotion of generated artifacts;
|
||||||
|
- semantic knowledge of particular artifact types;
|
||||||
|
- per-artifact resume semantics within a successful or failed analyze stage;
|
||||||
|
- automatic dependency inference without `depends_on`.
|
||||||
|
|
||||||
|
Narratio should continue to orchestrate a fixed pipeline. The configurable part is the set of Scriptorium artifact invocations performed during the `analyze` stage.
|
||||||
|
|
||||||
|
## Current State
|
||||||
|
|
||||||
|
Narratio already has several relevant pieces in place:
|
||||||
|
|
||||||
|
- `pipeline.scriptorium.artifacts` is modeled as a map of artifact definitions.
|
||||||
|
- The Scriptorium adapter already accepts generic run/render requests.
|
||||||
|
- The artifact resolver already understands canonical artifact source IDs.
|
||||||
|
- The `analyze` stage already resolves inputs, optionally runs render-debug, invokes Scriptorium, verifies output, and records metadata.
|
||||||
|
|
||||||
|
The main limitation is that `analyze` currently treats `session_recap` as the only executable artifact and rejects other enabled artifact definitions.
|
||||||
|
|
||||||
|
## Target Architecture
|
||||||
|
|
||||||
|
### Runtime Artifact Catalog
|
||||||
|
|
||||||
|
Introduce a per-run artifact catalog that tracks built-in artifacts and configured artifacts.
|
||||||
|
|
||||||
|
Conceptually:
|
||||||
|
|
||||||
|
```text
|
||||||
|
ArtifactCatalog
|
||||||
|
├── built-in artifacts
|
||||||
|
│ ├── narratio.transcript.merged
|
||||||
|
│ ├── narratio.transcript.polished
|
||||||
|
│ ├── narratio.transcript.full
|
||||||
|
│ ├── narratio.transcript.trimmed
|
||||||
|
│ └── narratio.bounds.session
|
||||||
|
│
|
||||||
|
└── configured artifacts
|
||||||
|
├── narratio.artifact.session_recap
|
||||||
|
├── narratio.artifact.player_handout
|
||||||
|
└── narratio.artifact.npc_summary
|
||||||
|
```
|
||||||
|
|
||||||
|
The catalog should distinguish between three states:
|
||||||
|
|
||||||
|
```text
|
||||||
|
planned valid configured or built-in artifact known to Narratio
|
||||||
|
available artifact has been produced or otherwise resolved
|
||||||
|
executable configured artifact selected for execution in this analyze run
|
||||||
|
```
|
||||||
|
|
||||||
|
Configured artifacts can be planned without being executable. This distinction is important for disabled artifacts and for `--artifacts` filtering.
|
||||||
|
|
||||||
|
### Configured Artifact Source IDs
|
||||||
|
|
||||||
|
Configured artifact keys map directly to source IDs:
|
||||||
|
|
||||||
|
```text
|
||||||
|
pipeline.scriptorium.artifacts.<name>
|
||||||
|
→ narratio.artifact.<name>
|
||||||
|
```
|
||||||
|
|
||||||
|
`session_recap` should no longer be a special built-in analyze artifact. Instead, it is just a conventional configured artifact key:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
scriptorium:
|
||||||
|
artifacts:
|
||||||
|
session_recap:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd_session.session_recap
|
||||||
|
output_path: artifacts/session_recap.md
|
||||||
|
```
|
||||||
|
|
||||||
|
`narratio.artifact.session_recap` remains valid only because `session_recap` is configured.
|
||||||
|
|
||||||
|
### Artifact Output Directory
|
||||||
|
|
||||||
|
Add an internal default artifact output directory, initially:
|
||||||
|
|
||||||
|
```text
|
||||||
|
artifacts
|
||||||
|
```
|
||||||
|
|
||||||
|
This default should live in `internal/config/defaults.go` or the existing equivalent defaults location.
|
||||||
|
|
||||||
|
For the initial implementation:
|
||||||
|
|
||||||
|
- expose no public config knob for the artifact output directory;
|
||||||
|
- require each configured artifact to provide an explicit `output_path`;
|
||||||
|
- validate that each configured artifact `output_path` is run-relative;
|
||||||
|
- validate that each configured artifact `output_path` is under the internal artifact output directory;
|
||||||
|
- reject output paths that escape the run workspace or use path traversal.
|
||||||
|
|
||||||
|
This preserves future configurability without forcing Narratio to guess output extensions or formats now.
|
||||||
|
|
||||||
|
### Enabled, Disabled, and Selected Artifacts
|
||||||
|
|
||||||
|
Configured artifacts should have three distinct execution states:
|
||||||
|
|
||||||
|
```text
|
||||||
|
enabled by config artifact has enabled: true
|
||||||
|
selected for execution artifact remains executable after --artifacts filtering
|
||||||
|
disabled for execution artifact is not executable, but may be resolvable from disk
|
||||||
|
```
|
||||||
|
|
||||||
|
Without `--artifacts`, all configured artifacts with `enabled: true` are selected for execution.
|
||||||
|
|
||||||
|
With `--artifacts`, only the named artifacts are selected for execution. All other configured artifacts are treated as disabled for the current analyze invocation, regardless of their configured `enabled` value.
|
||||||
|
|
||||||
|
Disabled artifacts may still be resolved as inputs if their configured `output_path` exists on disk and passes validation.
|
||||||
|
|
||||||
|
### Disabled Artifact Resolution
|
||||||
|
|
||||||
|
If artifact `B` references artifact `A`, and `A` is disabled for execution, Narratio should attempt to resolve `A` from disk.
|
||||||
|
|
||||||
|
This should succeed only when:
|
||||||
|
|
||||||
|
1. `A` is defined in `pipeline.scriptorium.artifacts`;
|
||||||
|
2. `A` has a valid `output_path`;
|
||||||
|
3. the output path exists in the current run workspace;
|
||||||
|
4. the output is non-empty, or otherwise passes any available artifact-specific validation.
|
||||||
|
|
||||||
|
The resolved provenance should make the source clear, for example:
|
||||||
|
|
||||||
|
```text
|
||||||
|
filesystem.disabled_artifact_output
|
||||||
|
```
|
||||||
|
|
||||||
|
If the file does not exist or fails validation, the dependent artifact should fail before invoking Scriptorium.
|
||||||
|
|
||||||
|
Example error wording:
|
||||||
|
|
||||||
|
```text
|
||||||
|
artifact player_handout requires narratio.artifact.session_recap, but session_recap is disabled for execution and artifacts/session_recap.md does not exist
|
||||||
|
```
|
||||||
|
|
||||||
|
### Explicit Dependencies
|
||||||
|
|
||||||
|
Artifact-to-artifact references require explicit `depends_on` entries.
|
||||||
|
|
||||||
|
If artifact `B` has an input source of `narratio.artifact.A`, then `B.depends_on` must include `A`.
|
||||||
|
|
||||||
|
This should fail:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
scriptorium:
|
||||||
|
artifacts:
|
||||||
|
player_handout:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd_session.player_handout
|
||||||
|
output_path: artifacts/player_handout.md
|
||||||
|
inputs:
|
||||||
|
recap:
|
||||||
|
source: narratio.artifact.session_recap
|
||||||
|
required: true
|
||||||
|
```
|
||||||
|
|
||||||
|
This should pass:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
scriptorium:
|
||||||
|
artifacts:
|
||||||
|
player_handout:
|
||||||
|
enabled: true
|
||||||
|
depends_on:
|
||||||
|
- session_recap
|
||||||
|
prompt_id: dnd_session.player_handout
|
||||||
|
output_path: artifacts/player_handout.md
|
||||||
|
inputs:
|
||||||
|
recap:
|
||||||
|
source: narratio.artifact.session_recap
|
||||||
|
required: true
|
||||||
|
```
|
||||||
|
|
||||||
|
`depends_on` values refer to configured artifact keys, not full source IDs.
|
||||||
|
|
||||||
|
Dependency validation should fail on:
|
||||||
|
|
||||||
|
- references to unknown artifact keys;
|
||||||
|
- missing `depends_on` entries for artifact-to-artifact input references;
|
||||||
|
- self-dependencies;
|
||||||
|
- dependency cycles among executable artifacts.
|
||||||
|
|
||||||
|
Dependencies on disabled artifacts are permitted, but the disabled dependency must resolve from disk before the dependent artifact runs.
|
||||||
|
|
||||||
|
### Execution Order
|
||||||
|
|
||||||
|
The analyze stage should execute selected artifacts in dependency order.
|
||||||
|
|
||||||
|
Rules:
|
||||||
|
|
||||||
|
- selected artifacts are executable;
|
||||||
|
- disabled artifacts are never executed;
|
||||||
|
- selected artifacts may depend on other selected artifacts;
|
||||||
|
- selected artifacts may depend on disabled artifacts if those disabled artifacts resolve from disk;
|
||||||
|
- independent selected artifacts run in deterministic sorted-name order.
|
||||||
|
|
||||||
|
Use topological sorting over selected artifacts, while validating dependency references across the full configured artifact set.
|
||||||
|
|
||||||
|
### Input Resolution
|
||||||
|
|
||||||
|
Input resolution should use the artifact catalog and existing artifact resolver behavior.
|
||||||
|
|
||||||
|
For each configured artifact input:
|
||||||
|
|
||||||
|
- built-in sources resolve through existing resolver behavior;
|
||||||
|
- `previous_session_artifact` preserves existing behavior;
|
||||||
|
- `narratio.artifact.<name>` resolves through the runtime artifact catalog;
|
||||||
|
- selected dependencies resolve after being produced earlier in the same analyze execution;
|
||||||
|
- disabled dependencies resolve from their configured output path on disk;
|
||||||
|
- optional missing inputs are omitted;
|
||||||
|
- required missing inputs fail before Scriptorium is invoked.
|
||||||
|
|
||||||
|
### Analyze Stage Generalization
|
||||||
|
|
||||||
|
The `analyze` stage should become the generic Scriptorium artifact stage.
|
||||||
|
|
||||||
|
High-level flow:
|
||||||
|
|
||||||
|
1. Load configured Scriptorium artifacts.
|
||||||
|
2. Apply the `--artifacts` filter, if present.
|
||||||
|
3. If no artifacts are selected for execution, return success metadata with `skipped=true`.
|
||||||
|
4. Build the runtime artifact catalog.
|
||||||
|
5. Validate artifact names, output paths, source IDs, dependencies, selected artifacts, and required fields.
|
||||||
|
6. Resolve any disabled dependencies that are required by selected artifacts.
|
||||||
|
7. Sort selected artifacts by dependency order.
|
||||||
|
8. For each selected artifact:
|
||||||
|
- resolve configured inputs;
|
||||||
|
- build the Scriptorium run request;
|
||||||
|
- optionally run Scriptorium render-debug;
|
||||||
|
- run Scriptorium;
|
||||||
|
- fail on validation-failed result;
|
||||||
|
- verify the output exists and is non-empty;
|
||||||
|
- record artifact output metadata;
|
||||||
|
- register `narratio.artifact.<name>` as available in the catalog.
|
||||||
|
9. Return aggregate analyze-stage metadata containing all generated and reused artifacts relevant to the run.
|
||||||
|
|
||||||
|
The Scriptorium adapter should remain generic. It should not decide which artifacts run, how dependencies work, or how artifacts are registered.
|
||||||
|
|
||||||
|
### Manifest and Metadata
|
||||||
|
|
||||||
|
The manifest should remain stage-oriented.
|
||||||
|
|
||||||
|
This means:
|
||||||
|
|
||||||
|
- `analyze` succeeds or fails as a full stage;
|
||||||
|
- if `analyze` has already succeeded and the user does not force it, the runner skips it as a full stage;
|
||||||
|
- Narratio should not implement per-artifact resume in the first version.
|
||||||
|
|
||||||
|
However, analyze-stage metadata should still record artifact outputs for provenance and future resolution.
|
||||||
|
|
||||||
|
Recommended metadata shape:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"skipped": false,
|
||||||
|
"artifacts": [
|
||||||
|
{
|
||||||
|
"name": "session_recap",
|
||||||
|
"source_id": "narratio.artifact.session_recap",
|
||||||
|
"output_kind": "scriptorium_artifact",
|
||||||
|
"path": "artifacts/session_recap.md",
|
||||||
|
"prompt_id": "dnd_session.session_recap",
|
||||||
|
"profile_id": "local-gemma-31b",
|
||||||
|
"provenance": "generated.current_analyze_run"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"name": "player_handout",
|
||||||
|
"source_id": "narratio.artifact.player_handout",
|
||||||
|
"output_kind": "scriptorium_artifact",
|
||||||
|
"path": "artifacts/player_handout.md",
|
||||||
|
"prompt_id": "dnd_session.player_handout",
|
||||||
|
"profile_id": "local-gemma-31b",
|
||||||
|
"provenance": "generated.current_analyze_run"
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"reused_artifacts": [
|
||||||
|
{
|
||||||
|
"name": "session_recap",
|
||||||
|
"source_id": "narratio.artifact.session_recap",
|
||||||
|
"path": "artifacts/session_recap.md",
|
||||||
|
"provenance": "filesystem.disabled_artifact_output"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
The exact struct can differ from this example, but it should preserve:
|
||||||
|
|
||||||
|
- artifact name;
|
||||||
|
- canonical source ID;
|
||||||
|
- output path;
|
||||||
|
- prompt/profile provenance for generated artifacts;
|
||||||
|
- reused-vs-generated provenance.
|
||||||
|
|
||||||
|
### Resume and Force Behavior
|
||||||
|
|
||||||
|
Keep resume behavior stage-level.
|
||||||
|
|
||||||
|
Recommended semantics:
|
||||||
|
|
||||||
|
```text
|
||||||
|
No --force, analyze already succeeded:
|
||||||
|
runner skips analyze, regardless of --artifacts.
|
||||||
|
|
||||||
|
--force, no --artifacts:
|
||||||
|
analyze regenerates all configured artifacts with enabled: true.
|
||||||
|
|
||||||
|
--force --artifacts player_handout:
|
||||||
|
analyze treats only player_handout as executable.
|
||||||
|
all other configured artifacts are disabled for execution.
|
||||||
|
disabled dependencies may be reused from disk.
|
||||||
|
|
||||||
|
--artifacts player_handout on a not-yet-completed analyze stage:
|
||||||
|
analyze runs only player_handout.
|
||||||
|
disabled dependencies may be reused from disk.
|
||||||
|
```
|
||||||
|
|
||||||
|
`--artifacts` should not imply `--force`. It is an execution filter, not a resume override.
|
||||||
|
|
||||||
|
### `--artifacts` CLI Flag
|
||||||
|
|
||||||
|
Add an `--artifacts` flag to commands that can execute or resume the analyze stage.
|
||||||
|
|
||||||
|
The flag should accept one or more configured artifact names. Internally, normalize values to a set of artifact keys.
|
||||||
|
|
||||||
|
Recommended behavior:
|
||||||
|
|
||||||
|
- validate all requested artifact names against `pipeline.scriptorium.artifacts`;
|
||||||
|
- reject unknown artifact names before running stages;
|
||||||
|
- treat requested artifacts as the only executable artifacts for the analyze stage;
|
||||||
|
- treat all other configured artifacts as disabled for execution;
|
||||||
|
- allow disabled artifacts to satisfy dependencies from disk as described above;
|
||||||
|
- if `--artifacts` is used while executing a stage other than `analyze`, either reject it or ignore it with a clear validation error. Prefer rejection.
|
||||||
|
|
||||||
|
The exact CLI parsing style can follow Narratio's existing conventions. Both comma-separated and repeatable values are acceptable if the CLI package supports them cleanly, but the internal representation should be a set of artifact keys.
|
||||||
|
|
||||||
|
### Archive Behavior
|
||||||
|
|
||||||
|
Do not automatically archive every generated artifact.
|
||||||
|
|
||||||
|
Artifact generation and archive promotion should remain separate concerns. Operators should continue to use `archive.promote_artifacts` to decide which generated files should be promoted or uploaded.
|
||||||
|
|
||||||
|
Example:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
archive:
|
||||||
|
promote_artifacts:
|
||||||
|
- from: artifacts/session_recap.md
|
||||||
|
to: artifacts/session_recap.md
|
||||||
|
required: true
|
||||||
|
- from: artifacts/player_handout.md
|
||||||
|
to: artifacts/player_handout.md
|
||||||
|
required: false
|
||||||
|
```
|
||||||
|
|
||||||
|
A later enhancement may add opt-in automatic promotion of configured artifacts, but explicit promotion should remain the default.
|
||||||
|
|
||||||
|
## Implementation Plan
|
||||||
|
|
||||||
|
### Phase 1: Config Model and Defaults
|
||||||
|
|
||||||
|
Add or update the configured artifact model to include:
|
||||||
|
|
||||||
|
- `enabled`;
|
||||||
|
- `depends_on`;
|
||||||
|
- `prompt_id`;
|
||||||
|
- `profile_id`;
|
||||||
|
- `output_path`;
|
||||||
|
- `timeout`;
|
||||||
|
- `render_debug`;
|
||||||
|
- `inputs`;
|
||||||
|
- `vars`.
|
||||||
|
|
||||||
|
Add an internal default artifact output directory in `internal/config/defaults.go`, initially set to `artifacts`.
|
||||||
|
|
||||||
|
Validation rules:
|
||||||
|
|
||||||
|
- artifact names must match a conservative identifier pattern such as `^[a-z][a-z0-9_]*$`;
|
||||||
|
- selected/executable artifacts require `prompt_id` and `output_path`;
|
||||||
|
- configured artifacts that may be referenced while disabled require `output_path`;
|
||||||
|
- configured artifact output paths must be run-relative;
|
||||||
|
- configured artifact output paths must live under the internal artifact output directory;
|
||||||
|
- configured artifact output paths must not escape the run workspace;
|
||||||
|
- `narratio.artifact.<name>` input sources must refer to configured artifact keys;
|
||||||
|
- any `narratio.artifact.<name>` input source must have a matching `depends_on` entry;
|
||||||
|
- `depends_on` entries must refer to configured artifact keys;
|
||||||
|
- dependencies must not contain self-references or executable cycles;
|
||||||
|
- input names and var names must remain compatible with the Scriptorium adapter's validation rules;
|
||||||
|
- unknown YAML fields must continue to fail strict decode.
|
||||||
|
|
||||||
|
Tests:
|
||||||
|
|
||||||
|
- valid single configured artifact;
|
||||||
|
- valid multiple independent artifacts;
|
||||||
|
- valid artifact-to-artifact dependency;
|
||||||
|
- valid dependency on disabled artifact with output path;
|
||||||
|
- invalid artifact name;
|
||||||
|
- missing required fields;
|
||||||
|
- output path outside `artifacts/`;
|
||||||
|
- dependency on missing artifact;
|
||||||
|
- missing `depends_on` for artifact input source;
|
||||||
|
- self-dependency;
|
||||||
|
- cycle detection;
|
||||||
|
- typo in `narratio.artifact.<name>` source;
|
||||||
|
- unknown YAML fields still fail strict decode.
|
||||||
|
|
||||||
|
### Phase 2: CLI Filtering
|
||||||
|
|
||||||
|
Add the `--artifacts` flag and carry the selected artifact set into the run execution options.
|
||||||
|
|
||||||
|
Implementation notes:
|
||||||
|
|
||||||
|
- parse values according to existing CLI conventions;
|
||||||
|
- normalize to artifact key strings;
|
||||||
|
- validate against configured artifact definitions after config load;
|
||||||
|
- make the selected set available to the analyze stage;
|
||||||
|
- reject use with commands or stages where analyze cannot run.
|
||||||
|
|
||||||
|
Tests:
|
||||||
|
|
||||||
|
- no `--artifacts` means all enabled artifacts are selected;
|
||||||
|
- one requested artifact is selected;
|
||||||
|
- multiple requested artifacts are selected;
|
||||||
|
- unknown requested artifact fails;
|
||||||
|
- `--artifacts` does not imply `--force`;
|
||||||
|
- `--artifacts` with already-succeeded analyze stage is skipped unless forced;
|
||||||
|
- `--artifacts` on unsupported stage command fails clearly.
|
||||||
|
|
||||||
|
### Phase 3: Runtime Artifact Catalog
|
||||||
|
|
||||||
|
Introduce an internal artifact catalog abstraction.
|
||||||
|
|
||||||
|
Responsibilities:
|
||||||
|
|
||||||
|
- register built-in artifact definitions;
|
||||||
|
- register configured artifact definitions;
|
||||||
|
- map configured artifact keys to `narratio.artifact.<name>` IDs;
|
||||||
|
- track planned, available, and executable artifact states;
|
||||||
|
- expose lookup by canonical source ID;
|
||||||
|
- record generated provenance;
|
||||||
|
- record disabled-from-disk provenance.
|
||||||
|
|
||||||
|
Keep the catalog narrow. It should not execute Scriptorium and should not understand prompt semantics.
|
||||||
|
|
||||||
|
Tests:
|
||||||
|
|
||||||
|
- built-in source lookup;
|
||||||
|
- configured source registration;
|
||||||
|
- duplicate/conflicting source handling;
|
||||||
|
- planned but unavailable artifact lookup;
|
||||||
|
- selected artifact state;
|
||||||
|
- disabled artifact state;
|
||||||
|
- registering an artifact as available after generation;
|
||||||
|
- registering a disabled artifact as available from disk;
|
||||||
|
- resolving a configured artifact from analyze metadata if that behavior is implemented.
|
||||||
|
|
||||||
|
### Phase 4: Resolver Integration
|
||||||
|
|
||||||
|
Update artifact resolution so configured artifact IDs are resolved through the runtime catalog.
|
||||||
|
|
||||||
|
Resolution behavior:
|
||||||
|
|
||||||
|
- built-in sources continue using existing resolver behavior;
|
||||||
|
- configured artifact sources resolve from catalog availability/provenance;
|
||||||
|
- selected configured artifacts become available after generation;
|
||||||
|
- disabled configured artifacts may become available from disk;
|
||||||
|
- missing optional configured artifact inputs are omitted;
|
||||||
|
- missing required configured artifact inputs fail clearly.
|
||||||
|
|
||||||
|
Tests:
|
||||||
|
|
||||||
|
- configured artifact consumes a built-in transcript source;
|
||||||
|
- configured artifact consumes another configured artifact produced earlier in the same analyze run;
|
||||||
|
- configured artifact consumes a disabled artifact resolved from disk;
|
||||||
|
- required disabled artifact missing on disk fails;
|
||||||
|
- required configured artifact missing fails;
|
||||||
|
- optional missing configured artifact is omitted;
|
||||||
|
- reused artifact provenance is recorded distinctly from generated artifact provenance.
|
||||||
|
|
||||||
|
### Phase 5: Analyze Stage Generalization
|
||||||
|
|
||||||
|
Refactor `analyze` to execute selected configured artifacts.
|
||||||
|
|
||||||
|
Implementation notes:
|
||||||
|
|
||||||
|
- remove the hard-coded `session_recap` selection path;
|
||||||
|
- remove the hard-coded rejection of non-`session_recap` artifacts;
|
||||||
|
- preserve skip behavior when Scriptorium config is absent or no artifacts are selected;
|
||||||
|
- build the runtime artifact catalog;
|
||||||
|
- apply `--artifacts` filtering;
|
||||||
|
- validate selected artifacts and their dependencies;
|
||||||
|
- pre-resolve disabled dependencies from disk where required;
|
||||||
|
- compute deterministic dependency order;
|
||||||
|
- execute selected artifacts one at a time in dependency order;
|
||||||
|
- keep render-debug behavior at global and artifact levels;
|
||||||
|
- keep Scriptorium adapter invocation generic;
|
||||||
|
- after each successful run, register the artifact as available in the catalog;
|
||||||
|
- aggregate generated and reused artifact metadata.
|
||||||
|
|
||||||
|
Tests:
|
||||||
|
|
||||||
|
- no Scriptorium config skips;
|
||||||
|
- empty artifact map skips;
|
||||||
|
- no selected artifacts skips;
|
||||||
|
- disabled artifacts do not run;
|
||||||
|
- one selected artifact runs;
|
||||||
|
- multiple independent artifacts run in deterministic order;
|
||||||
|
- dependent selected artifact receives prior selected artifact as input;
|
||||||
|
- dependent selected artifact receives disabled-from-disk artifact as input;
|
||||||
|
- render-debug works for configured artifacts;
|
||||||
|
- Scriptorium validation failure fails the stage;
|
||||||
|
- missing required input fails the stage;
|
||||||
|
- successful outputs are non-empty and recorded;
|
||||||
|
- artifact filter executes only requested artifacts.
|
||||||
|
|
||||||
|
### Phase 6: Manifest and Stage Metadata
|
||||||
|
|
||||||
|
Update analyze-stage metadata and manifest output recording to support dynamic configured artifacts.
|
||||||
|
|
||||||
|
Recommended behavior:
|
||||||
|
|
||||||
|
- every generated configured artifact gets `source_id: narratio.artifact.<name>`;
|
||||||
|
- every generated configured artifact gets a generic output kind such as `scriptorium_artifact`;
|
||||||
|
- reused disabled artifacts are recorded separately from generated artifacts;
|
||||||
|
- metadata is sufficient for debugging, provenance, and future resolver support;
|
||||||
|
- metadata does not create per-artifact resume semantics.
|
||||||
|
|
||||||
|
Because this is a pre-release hard cutover, do not preserve a special legacy `session_recap` output kind unless a current internal test or archive path still requires it temporarily. Prefer updating tests and examples to treat `session_recap` as an ordinary configured artifact.
|
||||||
|
|
||||||
|
Tests:
|
||||||
|
|
||||||
|
- metadata records one generated configured artifact;
|
||||||
|
- metadata records multiple generated configured artifacts;
|
||||||
|
- metadata records reused disabled artifact provenance;
|
||||||
|
- `session_recap` is recorded as a normal configured artifact;
|
||||||
|
- manifest still treats `analyze` as a single succeeded or failed stage;
|
||||||
|
- runner skip behavior remains stage-level.
|
||||||
|
|
||||||
|
### Phase 7: Archive and Promotion Review
|
||||||
|
|
||||||
|
Review archive behavior after dynamic artifacts are recorded.
|
||||||
|
|
||||||
|
Implementation notes:
|
||||||
|
|
||||||
|
- do not automatically promote every configured artifact;
|
||||||
|
- keep `archive.promote_artifacts` explicit;
|
||||||
|
- update default or example promotion rules to use configured `session_recap` output path;
|
||||||
|
- ensure required promotion rules fail clearly when selected artifact generation did not produce a required file.
|
||||||
|
|
||||||
|
Tests:
|
||||||
|
|
||||||
|
- generated artifact can be promoted by explicit archive rule;
|
||||||
|
- required archive promotion fails if selected artifact was not generated and no file exists;
|
||||||
|
- optional archive promotion skips cleanly if file is absent;
|
||||||
|
- hard cutover does not rely on hard-coded `session_recap` generation.
|
||||||
|
|
||||||
|
### Phase 8: Documentation and Examples
|
||||||
|
|
||||||
|
Status: complete.
|
||||||
|
|
||||||
|
Update documentation after the implementation is complete.
|
||||||
|
|
||||||
|
Recommended documentation changes:
|
||||||
|
|
||||||
|
- update `docs/config.md` with the generalized artifact configuration model;
|
||||||
|
- update `docs/internal/artifacts.md` to describe the runtime artifact catalog;
|
||||||
|
- update `docs/stages/analyze.md` to describe generic Scriptorium artifact generation;
|
||||||
|
- update Scriptorium integration docs only if the adapter contract changes;
|
||||||
|
- update full annotated pipeline examples;
|
||||||
|
- add at least one example with multiple artifacts and one dependency;
|
||||||
|
- document `--artifacts` behavior and its relationship to `--force`;
|
||||||
|
- remove documentation stating that only `session_recap` is supported.
|
||||||
|
|
||||||
|
Documentation should make clear that:
|
||||||
|
|
||||||
|
- configured artifact source IDs use `narratio.artifact.<name>`;
|
||||||
|
- `depends_on` uses artifact keys, not full source IDs;
|
||||||
|
- artifact-to-artifact source references require explicit `depends_on`;
|
||||||
|
- disabled artifacts can be reused from disk when required by selected artifacts;
|
||||||
|
- `--artifacts` filters execution but does not imply `--force`;
|
||||||
|
- archive promotion remains explicit;
|
||||||
|
- per-artifact resume is not part of the initial implementation.
|
||||||
|
|
||||||
|
## Migration Strategy
|
||||||
|
|
||||||
|
Because Narratio is pre-release, perform a hard cutover.
|
||||||
|
|
||||||
|
Required changes:
|
||||||
|
|
||||||
|
1. Remove the hard-coded `session_recap` analyze behavior.
|
||||||
|
2. Require `session_recap` to be declared under `pipeline.scriptorium.artifacts.session_recap` if the operator wants a session recap.
|
||||||
|
3. Treat `narratio.artifact.session_recap` as valid only when `session_recap` is a configured artifact key.
|
||||||
|
4. Update config examples to show `session_recap` as a normal configured artifact.
|
||||||
|
5. Update tests to stop assuming that `session_recap` is a built-in analyze artifact.
|
||||||
|
6. Keep archive promotion explicit and path-based.
|
||||||
|
|
||||||
|
Example replacement config:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
config_path: /etc/scriptorium/config.yml
|
||||||
|
timeout: 10m
|
||||||
|
render_debug: false
|
||||||
|
artifacts:
|
||||||
|
session_recap:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd_session.session_recap
|
||||||
|
profile_id: local-gemma-31b
|
||||||
|
output_path: artifacts/session_recap.md
|
||||||
|
timeout: 20m
|
||||||
|
inputs:
|
||||||
|
transcript:
|
||||||
|
source: narratio.transcript.trimmed
|
||||||
|
required: true
|
||||||
|
prior_recap:
|
||||||
|
source: previous_session_artifact
|
||||||
|
artifact: artifacts/session_recap.md
|
||||||
|
required: false
|
||||||
|
vars:
|
||||||
|
artifact_title: Session Recap
|
||||||
|
```
|
||||||
|
|
||||||
|
## Acceptance Criteria
|
||||||
|
|
||||||
|
The feature is complete when:
|
||||||
|
|
||||||
|
- operators can define more than one enabled Scriptorium artifact in `pipeline.yml`;
|
||||||
|
- Narratio runs selected artifacts in deterministic dependency order;
|
||||||
|
- configured artifacts are addressable as `narratio.artifact.<name>`;
|
||||||
|
- one configured artifact can consume another configured artifact as an input;
|
||||||
|
- artifact-to-artifact input references require explicit `depends_on`;
|
||||||
|
- disabled artifacts can satisfy dependencies from existing on-disk outputs;
|
||||||
|
- missing required disabled artifacts fail clearly;
|
||||||
|
- optional missing inputs are omitted;
|
||||||
|
- `--artifacts` can selectively execute valid configured artifact names;
|
||||||
|
- `--artifacts` does not imply `--force`;
|
||||||
|
- render-debug behavior works for all configured artifacts;
|
||||||
|
- generated and reused artifacts are recorded in analyze-stage metadata;
|
||||||
|
- `session_recap` is no longer hard-coded and works as a normal configured artifact;
|
||||||
|
- archive promotion remains explicit;
|
||||||
|
- tests cover config validation, dependency sorting, disabled artifact resolution, resolver behavior, CLI filtering, analyze execution, archive interactions, and metadata.
|
||||||
|
|
||||||
|
## Suggested Implementation Order
|
||||||
|
|
||||||
|
1. Config model, defaults, and validation.
|
||||||
|
2. CLI parsing and propagation of `--artifacts` selection.
|
||||||
|
3. Runtime artifact catalog.
|
||||||
|
4. Resolver integration for configured artifacts.
|
||||||
|
5. Analyze stage generalization.
|
||||||
|
6. Stage metadata and manifest output recording.
|
||||||
|
7. Archive behavior review.
|
||||||
|
8. Documentation and examples.
|
||||||
|
|
||||||
|
This order keeps the most static pieces first, then moves into execution behavior once the configuration contract is explicit and well tested.
|
||||||
@@ -1,119 +0,0 @@
|
|||||||
# Archive Storage
|
|
||||||
|
|
||||||
This document describes implemented archive-stage publish behavior.
|
|
||||||
|
|
||||||
## S3 Paths
|
|
||||||
|
|
||||||
Session root:
|
|
||||||
|
|
||||||
`{root_prefix}/campaigns/{campaign}/sessions/{session_id}/`
|
|
||||||
|
|
||||||
Run prefix:
|
|
||||||
|
|
||||||
`{root_prefix}/campaigns/{campaign}/sessions/{session_id}/runs/{run_id}/`
|
|
||||||
|
|
||||||
## Scope
|
|
||||||
|
|
||||||
Implemented:
|
|
||||||
|
|
||||||
- archive uploads successful run records to remote object storage through the storage backend abstraction.
|
|
||||||
- archive uploads configured promoted outputs to session-level keys.
|
|
||||||
- archive uploads `current/manifest.json`.
|
|
||||||
- archive uploads `current/run_id.txt` last as the effective commit marker.
|
|
||||||
- optional post-archive local cleanup:
|
|
||||||
- `pipeline.spool.delete_audio_after_archive: true` removes only the run-scoped spool audio directory
|
|
||||||
- `pipeline.workspace.cleanup_after_archive: true` removes only the run-scoped local workdir
|
|
||||||
- tests use fake storage and do not require live S3.
|
|
||||||
|
|
||||||
Future work:
|
|
||||||
|
|
||||||
- `notify` stage behavior
|
|
||||||
- stale detection
|
|
||||||
- optional future source-audio upload mode
|
|
||||||
- additional artifact generation beyond current implemented set
|
|
||||||
|
|
||||||
## Prerequisites
|
|
||||||
|
|
||||||
Archive verifies these stages succeeded before upload:
|
|
||||||
|
|
||||||
- `prepare`
|
|
||||||
- `transcribe`
|
|
||||||
- `merge`
|
|
||||||
- `polish`
|
|
||||||
- `normalize`
|
|
||||||
- `trim`
|
|
||||||
- `analyze`
|
|
||||||
|
|
||||||
If any prerequisite is missing or not succeeded, archive fails and does not upload.
|
|
||||||
Failed or incomplete runs remain local only.
|
|
||||||
|
|
||||||
## Run Upload
|
|
||||||
|
|
||||||
Archive uploads existing files from the run workdir when present:
|
|
||||||
|
|
||||||
- `inputs/`
|
|
||||||
- `transcripts/`
|
|
||||||
- `artifacts/`
|
|
||||||
- `reports/` (optional)
|
|
||||||
- `config/`
|
|
||||||
- `logs/`
|
|
||||||
- `manifest.json`
|
|
||||||
|
|
||||||
Relative paths are preserved under `runs/{run_id}/`.
|
|
||||||
|
|
||||||
## Promotion Rules
|
|
||||||
|
|
||||||
Archive applies `archive.promote_artifacts` in config order.
|
|
||||||
|
|
||||||
Rule behavior:
|
|
||||||
|
|
||||||
- `from`: local workdir-relative source path
|
|
||||||
- `to`: session-root-relative destination key
|
|
||||||
- `required: true`: missing source fails archive
|
|
||||||
- `required: false`: missing source is skipped and recorded
|
|
||||||
|
|
||||||
Default promoted outputs:
|
|
||||||
|
|
||||||
- `transcripts/trimmed.json`
|
|
||||||
- `artifacts/session_recap.md`
|
|
||||||
|
|
||||||
## Current Pointers
|
|
||||||
|
|
||||||
Archive writes:
|
|
||||||
|
|
||||||
1. `current/manifest.json` (after run upload + promotions)
|
|
||||||
2. `current/run_id.txt` last
|
|
||||||
|
|
||||||
`current/run_id.txt` contains exactly:
|
|
||||||
|
|
||||||
- `{run_id}` plus trailing newline
|
|
||||||
|
|
||||||
Writing `current/run_id.txt` last makes it the effective commit marker for published session state.
|
|
||||||
|
|
||||||
If any required run upload, promotion upload, or current-manifest upload fails, archive returns failure and does not write `current/run_id.txt`.
|
|
||||||
Cleanup runs only after this commit-marker write has succeeded.
|
|
||||||
|
|
||||||
## Audio Upload Policy
|
|
||||||
|
|
||||||
Archive does not upload local `audio/` by default.
|
|
||||||
Original audio is expected at the session-level audio prefix and is not duplicated under `runs/{run_id}/`.
|
|
||||||
|
|
||||||
## Config Controls
|
|
||||||
|
|
||||||
- `archive.enabled: false` skips archive cleanly.
|
|
||||||
- `archive.upload_run: false` skips run upload cleanly.
|
|
||||||
- both skip cases also skip post-archive local cleanup.
|
|
||||||
|
|
||||||
## Metadata
|
|
||||||
|
|
||||||
Archive stage metadata includes non-secret upload context (for example):
|
|
||||||
|
|
||||||
- `s3_bucket`
|
|
||||||
- `s3_run_prefix`
|
|
||||||
- run upload counts/paths
|
|
||||||
- promoted upload counts/paths
|
|
||||||
- skipped optional promotions
|
|
||||||
- `current_manifest_key`
|
|
||||||
- `current_run_id_key`
|
|
||||||
- `current_pointer_written`
|
|
||||||
- `audio_upload_skipped`
|
|
||||||
@@ -1,84 +0,0 @@
|
|||||||
# S3 Audio Input
|
|
||||||
|
|
||||||
This document describes implemented S3 audio input behavior in `prepare`.
|
|
||||||
|
|
||||||
## Scope
|
|
||||||
|
|
||||||
Implemented:
|
|
||||||
|
|
||||||
- `prepare` can acquire source audio from S3 when `session.inputs.audio_s3.prefix` is configured.
|
|
||||||
- object listing and download go through the storage backend abstraction.
|
|
||||||
- tests use fake storage; no live S3 service is required for test runs.
|
|
||||||
|
|
||||||
Not implemented:
|
|
||||||
|
|
||||||
- uploads of failed runs
|
|
||||||
|
|
||||||
## Required Configuration
|
|
||||||
|
|
||||||
`pipeline.yml`:
|
|
||||||
|
|
||||||
- `storage.s3.bucket` must be set when S3 audio input is used.
|
|
||||||
- `storage.s3.root_prefix` defaults to `dnd`.
|
|
||||||
- `storage.s3.access_key_id_env` defaults to `OBJECT_STORAGE_KEY_ID`.
|
|
||||||
- `storage.s3.secret_access_key_env` defaults to `OBJECT_STORAGE_KEY`.
|
|
||||||
- `spool.root` defaults to `/var/spool/narratio`.
|
|
||||||
|
|
||||||
`session.yml`:
|
|
||||||
|
|
||||||
- configure `session.campaign` and `session.session_id`.
|
|
||||||
- configure `session.inputs.audio_s3.prefix` for S3 audio input.
|
|
||||||
- do not configure `inputs.audio_dir` or `inputs.audio_files` at the same time as `inputs.audio_s3`.
|
|
||||||
|
|
||||||
## Prefix Shape
|
|
||||||
|
|
||||||
Session S3 root:
|
|
||||||
|
|
||||||
`{root_prefix}/campaigns/{campaign}/sessions/{session_id}/`
|
|
||||||
|
|
||||||
Audio prefix:
|
|
||||||
|
|
||||||
`{session_root}/{audio_s3.prefix}`
|
|
||||||
|
|
||||||
Example:
|
|
||||||
|
|
||||||
`dnd/campaigns/forsaken/sessions/2026-04-19/audio/`
|
|
||||||
|
|
||||||
Audio files must already exist in S3 before running Narratio.
|
|
||||||
|
|
||||||
## Prepare Behavior
|
|
||||||
|
|
||||||
When `inputs.audio_s3.prefix` is configured, `prepare`:
|
|
||||||
|
|
||||||
1. lists objects under the computed S3 audio prefix
|
|
||||||
2. filters to `.flac` objects
|
|
||||||
3. fails when no `.flac` objects are found
|
|
||||||
4. downloads selected objects to spool audio:
|
|
||||||
- `{spool.root}/{campaign}/{session_id}/{run_id}/audio/`
|
|
||||||
5. materializes audio into workdir audio:
|
|
||||||
- `{workspace.root}/work/{campaign}/{session_id}/runs/{run_id}/audio/`
|
|
||||||
6. records input provenance in the manifest (bucket, key, metadata, local paths, checksum)
|
|
||||||
|
|
||||||
Notes:
|
|
||||||
|
|
||||||
- `.flac` filtering is case-insensitive.
|
|
||||||
- ETag is recorded as provider metadata only and is not treated as a checksum.
|
|
||||||
|
|
||||||
## Local Audio Development
|
|
||||||
|
|
||||||
Local audio workflows remain supported:
|
|
||||||
|
|
||||||
- `inputs.audio_dir`
|
|
||||||
- `inputs.audio_files`
|
|
||||||
|
|
||||||
These options are mutually exclusive with `inputs.audio_s3`.
|
|
||||||
|
|
||||||
## Archive Boundary
|
|
||||||
|
|
||||||
Current archive behavior relevant to S3 audio input:
|
|
||||||
|
|
||||||
- successful runs are uploaded by archive under `runs/{run_id}/`
|
|
||||||
- configured promotions are uploaded to session-level destinations
|
|
||||||
- `current/manifest.json` and `current/run_id.txt` are published
|
|
||||||
- local source audio is not re-uploaded by default
|
|
||||||
- failed or incomplete runs are not uploaded
|
|
||||||
289
docs/troubleshooting.md
Normal file
289
docs/troubleshooting.md
Normal file
@@ -0,0 +1,289 @@
|
|||||||
|
# Troubleshooting
|
||||||
|
|
||||||
|
## Purpose
|
||||||
|
Canonical operator troubleshooting guide for recurring implemented Narratio failures.
|
||||||
|
|
||||||
|
## Config file discovery failure
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
- `run`, `plan`, `resume`, or `run-stage` fails with config/session not found.
|
||||||
|
|
||||||
|
Likely Cause:
|
||||||
|
- `pipeline.yml` or `session.yml` is missing from discovery paths.
|
||||||
|
- wrong working directory when relying on `./session.yml`.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pwd
|
||||||
|
ls -l ./session.yml
|
||||||
|
ls -l /usr/local/etc/narratio/pipeline.yml /etc/narratio/pipeline.yml
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe Fix:
|
||||||
|
- pass explicit `--config` and `--session`.
|
||||||
|
- or place files in documented discovery paths.
|
||||||
|
|
||||||
|
Links:
|
||||||
|
- [docs/config.md](./config.md)
|
||||||
|
- [docs/cli.md](./cli.md)
|
||||||
|
|
||||||
|
## Session template rendering failure
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
- load fails with unresolved placeholder or `session_id` mismatch.
|
||||||
|
|
||||||
|
Likely Cause:
|
||||||
|
- templated `session.yml` used without `--session-id`.
|
||||||
|
- rendered `session_id` differs from passed `--session-id`.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio plan --session ./session.yml --session-id 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe Fix:
|
||||||
|
- pass `--session-id` when template placeholders are present.
|
||||||
|
- ensure rendered `session_id` matches intended run session id.
|
||||||
|
|
||||||
|
Links:
|
||||||
|
- [docs/config.md](./config.md)
|
||||||
|
|
||||||
|
## Strict YAML decode or validation failure
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
- config load fails with unknown field or validation error.
|
||||||
|
|
||||||
|
Likely Cause:
|
||||||
|
- typo/stale field name.
|
||||||
|
- missing required fields or invalid constraints.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio plan --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe Fix:
|
||||||
|
- align fields/values to canonical config reference and examples.
|
||||||
|
|
||||||
|
Links:
|
||||||
|
- [docs/config.md](./config.md)
|
||||||
|
- [examples/](../examples/)
|
||||||
|
|
||||||
|
## `--artifacts` selection failure
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
- `run`/`resume`/`run-stage` fails with invalid or unknown artifact selection.
|
||||||
|
|
||||||
|
Likely Cause:
|
||||||
|
- `--artifacts` contains blank names or unknown artifact keys.
|
||||||
|
- `pipeline.scriptorium.artifacts` missing while using `--artifacts`.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --artifacts player_handout
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe Fix:
|
||||||
|
- use configured artifact keys only.
|
||||||
|
- ensure `pipeline.scriptorium.artifacts` is defined.
|
||||||
|
|
||||||
|
Links:
|
||||||
|
- [docs/cli.md](./cli.md)
|
||||||
|
- [docs/config.md](./config.md)
|
||||||
|
|
||||||
|
## `run-stage --artifacts` on non-analyze stage
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
- `run-stage` fails with `--artifacts is only supported for stage "analyze"`.
|
||||||
|
|
||||||
|
Likely Cause:
|
||||||
|
- `--artifacts` was used with a non-`analyze` stage.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --artifacts session_recap polish
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe Fix:
|
||||||
|
- use `--artifacts` only with `run-stage ... analyze`.
|
||||||
|
|
||||||
|
Links:
|
||||||
|
- [docs/cli.md](./cli.md)
|
||||||
|
|
||||||
|
## Configured artifact dependency/input validation failure
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
- config validation fails for `depends_on`, `narratio.artifact.<name>` source, or artifact output path.
|
||||||
|
|
||||||
|
Likely Cause:
|
||||||
|
- `narratio.artifact.<name>` source missing matching `depends_on` key.
|
||||||
|
- dependency references unknown artifact key.
|
||||||
|
- dependency self-reference or enabled dependency cycle.
|
||||||
|
- artifact output path missing/invalid/outside `artifacts/` root.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio plan --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe Fix:
|
||||||
|
- ensure artifact-to-artifact inputs have explicit `depends_on` entries using artifact keys.
|
||||||
|
- ensure referenced artifacts exist and define valid `output_path` values.
|
||||||
|
- keep output paths relative and under `artifacts/`.
|
||||||
|
|
||||||
|
Links:
|
||||||
|
- [docs/config.md](./config.md)
|
||||||
|
- [docs/internal/stage-analyze.md](./internal/stage-analyze.md)
|
||||||
|
|
||||||
|
## Required configured artifact input unavailable at analyze time
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
- analyze fails because configured input source is unavailable.
|
||||||
|
|
||||||
|
Likely Cause:
|
||||||
|
- required upstream configured artifact was not selected/executed this run.
|
||||||
|
- non-executable dependency output file is missing or invalid on disk.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio status --manifest /path/to/manifest.json
|
||||||
|
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 --artifacts player_handout analyze
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe Fix:
|
||||||
|
- run analyze with needed artifacts selected.
|
||||||
|
- or ensure dependency output file exists at configured path and is valid.
|
||||||
|
|
||||||
|
Links:
|
||||||
|
- [docs/operations.md](./operations.md)
|
||||||
|
- [docs/config.md](./config.md)
|
||||||
|
|
||||||
|
## Manifest/status path failure
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
- `status` fails because manifest path is missing, unreadable, or invalid.
|
||||||
|
|
||||||
|
Likely Cause:
|
||||||
|
- wrong manifest path.
|
||||||
|
- manifest removed after cleanup.
|
||||||
|
- `--manifest` omitted.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio status --manifest /path/to/manifest.json
|
||||||
|
ls -l /path/to/manifest.json
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe Fix:
|
||||||
|
- use manifest path printed by `run`, `resume`, or `run-stage`.
|
||||||
|
|
||||||
|
Links:
|
||||||
|
- [docs/cli.md](./cli.md)
|
||||||
|
- [docs/operations.md](./operations.md)
|
||||||
|
|
||||||
|
## Session lock conflict (`.lock`)
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
- run fails with lock conflict for session workdir.
|
||||||
|
|
||||||
|
Likely Cause:
|
||||||
|
- another Narratio process is running same session.
|
||||||
|
- stale lock from interrupted prior run.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ls -l {workspace.root}/work/{campaign}/{session_id}/.lock
|
||||||
|
cat {workspace.root}/work/{campaign}/{session_id}/.lock
|
||||||
|
ps aux | grep narratio
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe Fix:
|
||||||
|
- wait for active run to finish.
|
||||||
|
- if no process is active, remove only stale session `.lock` file.
|
||||||
|
|
||||||
|
Links:
|
||||||
|
- [docs/operations.md](./operations.md)
|
||||||
|
- [docs/internal/workspace.md](./internal/workspace.md)
|
||||||
|
|
||||||
|
## Secrets env-dir or credential-env failure
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
- startup fails loading secrets directory, or stage fails due to missing credential env vars.
|
||||||
|
|
||||||
|
Likely Cause:
|
||||||
|
- invalid `pipeline.secrets.env_dir` path/permissions.
|
||||||
|
- required credential env var unset/empty.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ls -la /path/to/secrets_dir
|
||||||
|
env | grep -E 'AUDITA|OBJECT_STORAGE|AWS|SCRIPTORIUM'
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe Fix:
|
||||||
|
- fix secrets directory and credential env vars.
|
||||||
|
- keep secret values out of YAML.
|
||||||
|
|
||||||
|
Links:
|
||||||
|
- [docs/config.md](./config.md)
|
||||||
|
|
||||||
|
## S3-audio prepare failure
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
- `prepare` fails in S3 mode (listing/downloading/no audio/backend error).
|
||||||
|
|
||||||
|
Likely Cause:
|
||||||
|
- wrong `session.inputs.audio_s3.prefix`.
|
||||||
|
- no `.flac` files at resolved prefix.
|
||||||
|
- invalid/missing object-store credentials or backend config.
|
||||||
|
- mixed local+S3 audio input config.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 prepare
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe Fix:
|
||||||
|
- configure exactly one audio source mode.
|
||||||
|
- verify `.flac` files and storage access.
|
||||||
|
|
||||||
|
Links:
|
||||||
|
- [docs/config.md](./config.md)
|
||||||
|
- [docs/operations.md](./operations.md)
|
||||||
|
|
||||||
|
## Archive promotion/current-pointer failure
|
||||||
|
|
||||||
|
Symptom:
|
||||||
|
- archive fails on required promotion source missing or pointer write failure.
|
||||||
|
|
||||||
|
Likely Cause:
|
||||||
|
- required promoted file absent (including analyze outputs not generated for this run).
|
||||||
|
- storage upload failed before `current/run_id.txt` commit marker write.
|
||||||
|
|
||||||
|
Diagnostics:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
narratio status --manifest /path/to/manifest.json
|
||||||
|
narratio run-stage --config /path/to/pipeline.yml --session /path/to/session.yml --session-id 2026-04-04 archive
|
||||||
|
```
|
||||||
|
|
||||||
|
Safe Fix:
|
||||||
|
- rerun or resume upstream stages to generate required files.
|
||||||
|
- adjust promotion rules to match files that must exist.
|
||||||
|
- retry after storage issue is resolved.
|
||||||
|
|
||||||
|
Links:
|
||||||
|
- [docs/operations.md](./operations.md)
|
||||||
|
- [docs/config.md](./config.md)
|
||||||
|
- [docs/internal/stage-archive.md](./internal/stage-archive.md)
|
||||||
@@ -1,26 +0,0 @@
|
|||||||
workspace:
|
|
||||||
root: ./tmp/narratio-workspace
|
|
||||||
|
|
||||||
whisperx:
|
|
||||||
transcribe_url: "https://transcription.example.com/transcribe"
|
|
||||||
|
|
||||||
seriatim:
|
|
||||||
binary: "seriatim"
|
|
||||||
timeout: "10m"
|
|
||||||
output_schema: "seriatim-intermediate"
|
|
||||||
coalesce_gap: 3.0
|
|
||||||
|
|
||||||
audita:
|
|
||||||
binary: "audita"
|
|
||||||
timeout: "3h"
|
|
||||||
base_url: "https://openrouter.ai/api/v1"
|
|
||||||
model: "openrouter/google/gemma-4-31b-it"
|
|
||||||
llm_api_key_env: "AUDITA_LLM_API_KEY"
|
|
||||||
modules: ["glossary", "homophones", "spoken_word", "grammar"]
|
|
||||||
output_schema: "audita-v1"
|
|
||||||
work_dir_retention: "auto"
|
|
||||||
total_llm_concurrency: 2
|
|
||||||
proposal_llm_concurrency: 1
|
|
||||||
validation_model: "openrouter/google/gemma-4-31b-it"
|
|
||||||
validation_llm_concurrency: 1
|
|
||||||
report: true
|
|
||||||
182
examples/pipeline.full.annotated.yml
Normal file
182
examples/pipeline.full.annotated.yml
Normal file
@@ -0,0 +1,182 @@
|
|||||||
|
# Full annotated pipeline example for implemented Narratio config fields.
|
||||||
|
# Values are safe placeholders and must be adapted per environment.
|
||||||
|
|
||||||
|
workspace:
|
||||||
|
# Optional: defaults to /var/lib/narratio.
|
||||||
|
root: /var/lib/narratio/workspace
|
||||||
|
# Optional: remove run-scoped workdir after successful archive commit.
|
||||||
|
cleanup_after_archive: false
|
||||||
|
|
||||||
|
# Optional: local secret file loader (directory of ENV_VAR_NAME files).
|
||||||
|
# secrets:
|
||||||
|
# env_dir: ./secrets
|
||||||
|
|
||||||
|
storage:
|
||||||
|
# Optional storage backend selector; use "s3" for archive + S3 audio workflows.
|
||||||
|
backend: s3
|
||||||
|
# Compatibility fields retained in schema.
|
||||||
|
bucket: ""
|
||||||
|
prefix: ""
|
||||||
|
s3:
|
||||||
|
# Required when using S3 audio or S3 archive uploads.
|
||||||
|
bucket: my-dnd-archive
|
||||||
|
# Optional; defaults to "dnd".
|
||||||
|
root_prefix: dnd
|
||||||
|
# Optional region/endpoint settings.
|
||||||
|
region: us-east-1
|
||||||
|
endpoint: ""
|
||||||
|
force_path_style: false
|
||||||
|
# Optional; defaults shown explicitly.
|
||||||
|
access_key_id_env: OBJECT_STORAGE_KEY_ID
|
||||||
|
secret_access_key_env: OBJECT_STORAGE_KEY
|
||||||
|
|
||||||
|
spool:
|
||||||
|
# Optional; defaults to /var/spool/narratio.
|
||||||
|
root: /var/spool/narratio
|
||||||
|
# Optional cleanup of run-scoped spool audio after successful archive commit.
|
||||||
|
delete_audio_after_archive: false
|
||||||
|
|
||||||
|
archive:
|
||||||
|
# Optional booleans; defaults are true.
|
||||||
|
enabled: true
|
||||||
|
upload_run: true
|
||||||
|
# Optional promotion rules; required files fail archive if missing.
|
||||||
|
promote_artifacts:
|
||||||
|
- from: transcripts/trimmed.json
|
||||||
|
to: transcripts/trimmed.json
|
||||||
|
required: true
|
||||||
|
- from: artifacts/session_recap.md
|
||||||
|
to: artifacts/session_recap.md
|
||||||
|
required: true
|
||||||
|
- from: artifacts/player_handout.md
|
||||||
|
to: artifacts/player_handout.md
|
||||||
|
required: false
|
||||||
|
|
||||||
|
whisperx:
|
||||||
|
# Required.
|
||||||
|
transcribe_url: "https://transcription.example.com/transcribe"
|
||||||
|
# Optional overrides; defaults shown explicitly.
|
||||||
|
language: en
|
||||||
|
timeout: 30m
|
||||||
|
retries: 3
|
||||||
|
retry_delay: 2s
|
||||||
|
concurrency: 2
|
||||||
|
|
||||||
|
seriatim:
|
||||||
|
# Optional overrides; defaults shown explicitly.
|
||||||
|
binary: seriatim
|
||||||
|
timeout: 10m
|
||||||
|
output_schema: seriatim-intermediate
|
||||||
|
coalesce_gap: 3.0
|
||||||
|
report: true
|
||||||
|
env:
|
||||||
|
# Optional advanced tuning; set only when needed.
|
||||||
|
overlap_word_run_gap: 1.0
|
||||||
|
overlap_word_run_reorder_window: 1.0
|
||||||
|
backchannel_max_duration: 2.0
|
||||||
|
filler_max_duration: 1.25
|
||||||
|
|
||||||
|
audita:
|
||||||
|
# Optional overrides; defaults shown explicitly where applicable.
|
||||||
|
binary: audita
|
||||||
|
timeout: 3h
|
||||||
|
llm_api_key_env: AUDITA_LLM_API_KEY
|
||||||
|
modules: [glossary, homophones, spoken_word, grammar]
|
||||||
|
base_url: ""
|
||||||
|
model: ""
|
||||||
|
total_llm_concurrency: 2
|
||||||
|
proposal_llm_concurrency: 1
|
||||||
|
validation_model: ""
|
||||||
|
validation_llm_concurrency: 1
|
||||||
|
transcript_description: ""
|
||||||
|
config_path: /usr/local/etc/audita/config.yml
|
||||||
|
output_schema: audita-v1
|
||||||
|
work_dir_retention: auto
|
||||||
|
report: true
|
||||||
|
|
||||||
|
normalize:
|
||||||
|
# Optional; defaults shown explicitly.
|
||||||
|
output_path: transcripts/normalized.json
|
||||||
|
output_schema: seriatim-intermediate
|
||||||
|
report: true
|
||||||
|
|
||||||
|
trim:
|
||||||
|
# Keep disabled unless bounds prompt integration is configured.
|
||||||
|
enabled: false
|
||||||
|
output_path: transcripts/trimmed.json
|
||||||
|
bounds:
|
||||||
|
prompt_id: dnd.session_bounds
|
||||||
|
profile_id: local-fast
|
||||||
|
transcript_input_name: transcript
|
||||||
|
output_path: reports/session_bounds.json
|
||||||
|
timeout: 10m
|
||||||
|
render_debug: false
|
||||||
|
render_output_path: reports/session_bounds.render.json
|
||||||
|
seriatim:
|
||||||
|
report: false
|
||||||
|
|
||||||
|
scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
config_path: /usr/local/etc/scriptorium/config.yml
|
||||||
|
timeout: 10m
|
||||||
|
render_debug: false
|
||||||
|
artifacts:
|
||||||
|
# Configured artifact keys map to source IDs narratio.artifact.<key>.
|
||||||
|
session_recap:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd.session_recap
|
||||||
|
profile_id: local-fast
|
||||||
|
output_path: artifacts/session_recap.md
|
||||||
|
timeout: 10m
|
||||||
|
inputs:
|
||||||
|
transcript:
|
||||||
|
source: narratio.transcript.trimmed
|
||||||
|
required: true
|
||||||
|
previous_recap:
|
||||||
|
source: previous_session_artifact
|
||||||
|
artifact: session_recap
|
||||||
|
path: ""
|
||||||
|
required: false
|
||||||
|
vars:
|
||||||
|
session_id: true
|
||||||
|
session_date: true
|
||||||
|
campaign_name: true
|
||||||
|
previous_session_id: true
|
||||||
|
output_kind: session_recap
|
||||||
|
|
||||||
|
# Example dependent artifact:
|
||||||
|
# - depends_on entries use artifact keys.
|
||||||
|
# - narratio.artifact.<key> sources require matching depends_on membership.
|
||||||
|
player_handout:
|
||||||
|
enabled: true
|
||||||
|
depends_on:
|
||||||
|
- session_recap
|
||||||
|
prompt_id: dnd.player_handout
|
||||||
|
profile_id: local-fast
|
||||||
|
output_path: artifacts/player_handout.md
|
||||||
|
timeout: 10m
|
||||||
|
inputs:
|
||||||
|
recap:
|
||||||
|
source: narratio.artifact.session_recap
|
||||||
|
required: true
|
||||||
|
transcript:
|
||||||
|
source: narratio.transcript.trimmed
|
||||||
|
required: true
|
||||||
|
vars:
|
||||||
|
session_id: true
|
||||||
|
campaign_name: true
|
||||||
|
output_kind: player_handout
|
||||||
|
|
||||||
|
analyzer:
|
||||||
|
# Optional adapter settings.
|
||||||
|
binary_path: ""
|
||||||
|
timeout: 2m
|
||||||
|
artifacts:
|
||||||
|
output_dir: ""
|
||||||
|
types: []
|
||||||
|
|
||||||
|
notification:
|
||||||
|
# Optional notification settings.
|
||||||
|
backend: ""
|
||||||
|
recipient: ""
|
||||||
|
timeout: 30s
|
||||||
@@ -1,55 +1,2 @@
|
|||||||
workspace:
|
|
||||||
root: ./tmp/narratio-workspace
|
|
||||||
cleanup_after_archive: false
|
|
||||||
|
|
||||||
storage:
|
|
||||||
backend: s3
|
|
||||||
s3:
|
|
||||||
bucket: "my-dnd-archive"
|
|
||||||
root_prefix: "dnd"
|
|
||||||
region: "us-east-1"
|
|
||||||
# Optional credential env-var names (defaulted when omitted):
|
|
||||||
# access_key_id_env: "OBJECT_STORAGE_KEY_ID"
|
|
||||||
# secret_access_key_env: "OBJECT_STORAGE_KEY"
|
|
||||||
|
|
||||||
spool:
|
|
||||||
root: "/var/spool/narratio"
|
|
||||||
delete_audio_after_archive: false
|
|
||||||
|
|
||||||
archive:
|
|
||||||
enabled: true
|
|
||||||
upload_run: true
|
|
||||||
|
|
||||||
whisperx:
|
whisperx:
|
||||||
transcribe_url: "https://transcription.example.com/transcribe"
|
transcribe_url: "https://transcription.example.com/transcribe"
|
||||||
|
|
||||||
# Optional. When omitted entirely, Narratio defaults to seriatim binary + runtime defaults.
|
|
||||||
seriatim: {}
|
|
||||||
|
|
||||||
# Optional runtime overrides. Model/provider can be owned by Audita runtime config.
|
|
||||||
audita:
|
|
||||||
config_path: "/usr/local/etc/audita/config.yml"
|
|
||||||
llm_api_key_env: "AUDITA_LLM_API_KEY"
|
|
||||||
|
|
||||||
# Optional Scriptorium integration for analyze artifacts.
|
|
||||||
scriptorium:
|
|
||||||
config_path: "/usr/local/etc/scriptorium/config.yml"
|
|
||||||
artifacts:
|
|
||||||
session_recap:
|
|
||||||
enabled: true
|
|
||||||
prompt_id: "dnd.session_recap"
|
|
||||||
output_path: "artifacts/session_recap.md"
|
|
||||||
inputs:
|
|
||||||
transcript:
|
|
||||||
source: "trimmed_transcript"
|
|
||||||
required: true
|
|
||||||
previous_recap:
|
|
||||||
source: "previous_session_artifact"
|
|
||||||
artifact: "session_recap"
|
|
||||||
required: false
|
|
||||||
vars:
|
|
||||||
session_id: true
|
|
||||||
session_date: true
|
|
||||||
campaign_name: true
|
|
||||||
previous_session_id: true
|
|
||||||
output_kind: "session_recap"
|
|
||||||
|
|||||||
116
examples/pipeline.production.yml
Normal file
116
examples/pipeline.production.yml
Normal file
@@ -0,0 +1,116 @@
|
|||||||
|
workspace:
|
||||||
|
root: /var/lib/narratio/workspace
|
||||||
|
cleanup_after_archive: true
|
||||||
|
|
||||||
|
storage:
|
||||||
|
backend: s3
|
||||||
|
s3:
|
||||||
|
bucket: my-dnd-archive
|
||||||
|
root_prefix: dnd
|
||||||
|
region: us-east-1
|
||||||
|
access_key_id_env: OBJECT_STORAGE_KEY_ID
|
||||||
|
secret_access_key_env: OBJECT_STORAGE_KEY
|
||||||
|
|
||||||
|
spool:
|
||||||
|
root: /var/spool/narratio
|
||||||
|
delete_audio_after_archive: true
|
||||||
|
|
||||||
|
archive:
|
||||||
|
enabled: true
|
||||||
|
upload_run: true
|
||||||
|
promote_artifacts:
|
||||||
|
- from: transcripts/trimmed.json
|
||||||
|
to: transcripts/trimmed.json
|
||||||
|
required: true
|
||||||
|
- from: artifacts/session_recap.md
|
||||||
|
to: artifacts/session_recap.md
|
||||||
|
required: true
|
||||||
|
- from: artifacts/player_handout.md
|
||||||
|
to: artifacts/player_handout.md
|
||||||
|
required: false
|
||||||
|
|
||||||
|
whisperx:
|
||||||
|
transcribe_url: "https://transcription.example.com/transcribe"
|
||||||
|
language: en
|
||||||
|
timeout: 45m
|
||||||
|
retries: 3
|
||||||
|
retry_delay: 3s
|
||||||
|
concurrency: 2
|
||||||
|
|
||||||
|
seriatim:
|
||||||
|
binary: seriatim
|
||||||
|
timeout: 10m
|
||||||
|
output_schema: seriatim-intermediate
|
||||||
|
coalesce_gap: 3.0
|
||||||
|
report: true
|
||||||
|
|
||||||
|
audita:
|
||||||
|
binary: audita
|
||||||
|
timeout: 3h
|
||||||
|
llm_api_key_env: AUDITA_LLM_API_KEY
|
||||||
|
modules: [glossary, homophones, spoken_word, grammar]
|
||||||
|
output_schema: audita-v1
|
||||||
|
work_dir_retention: auto
|
||||||
|
total_llm_concurrency: 2
|
||||||
|
proposal_llm_concurrency: 1
|
||||||
|
validation_llm_concurrency: 1
|
||||||
|
report: true
|
||||||
|
|
||||||
|
normalize:
|
||||||
|
output_path: transcripts/normalized.json
|
||||||
|
output_schema: seriatim-intermediate
|
||||||
|
report: true
|
||||||
|
|
||||||
|
trim:
|
||||||
|
enabled: false
|
||||||
|
|
||||||
|
scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
config_path: /usr/local/etc/scriptorium/config.yml
|
||||||
|
timeout: 10m
|
||||||
|
render_debug: false
|
||||||
|
artifacts:
|
||||||
|
session_recap:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd.session_recap
|
||||||
|
profile_id: local-fast
|
||||||
|
output_path: artifacts/session_recap.md
|
||||||
|
timeout: 10m
|
||||||
|
inputs:
|
||||||
|
transcript:
|
||||||
|
source: narratio.transcript.trimmed
|
||||||
|
required: true
|
||||||
|
previous_recap:
|
||||||
|
source: previous_session_artifact
|
||||||
|
artifact: session_recap
|
||||||
|
required: false
|
||||||
|
vars:
|
||||||
|
session_id: true
|
||||||
|
session_date: true
|
||||||
|
campaign_name: true
|
||||||
|
previous_session_id: true
|
||||||
|
output_kind: session_recap
|
||||||
|
player_handout:
|
||||||
|
enabled: true
|
||||||
|
depends_on:
|
||||||
|
- session_recap
|
||||||
|
prompt_id: dnd.player_handout
|
||||||
|
profile_id: local-fast
|
||||||
|
output_path: artifacts/player_handout.md
|
||||||
|
timeout: 10m
|
||||||
|
inputs:
|
||||||
|
recap:
|
||||||
|
source: narratio.artifact.session_recap
|
||||||
|
required: true
|
||||||
|
transcript:
|
||||||
|
source: narratio.transcript.trimmed
|
||||||
|
required: true
|
||||||
|
vars:
|
||||||
|
session_id: true
|
||||||
|
output_kind: player_handout
|
||||||
|
|
||||||
|
analyzer:
|
||||||
|
timeout: 2m
|
||||||
|
|
||||||
|
notification:
|
||||||
|
timeout: 30s
|
||||||
9
examples/session.local-audio.yml
Normal file
9
examples/session.local-audio.yml
Normal file
@@ -0,0 +1,9 @@
|
|||||||
|
session_id: 2026-05-03
|
||||||
|
campaign: sample-campaign
|
||||||
|
date: 2026-05-03
|
||||||
|
title: Sample Session
|
||||||
|
inputs:
|
||||||
|
audio_dir: ./audio
|
||||||
|
speakers_file: ./examples/speakers.yml
|
||||||
|
autocorrect_file: ./examples/autocorrect.yml
|
||||||
|
glossary_file: ./examples/glossary.yml
|
||||||
@@ -1,13 +0,0 @@
|
|||||||
session_id: 2026-05-03
|
|
||||||
campaign: sample-campaign
|
|
||||||
date: 2026-05-03
|
|
||||||
title: Sample Session
|
|
||||||
inputs:
|
|
||||||
audio_dir: ./audio
|
|
||||||
# Optional S3 input alternative. Do not configure with audio_dir/audio_files.
|
|
||||||
# Narratio prepare lists this prefix and downloads .flac files.
|
|
||||||
# audio_s3:
|
|
||||||
# prefix: "audio/"
|
|
||||||
speakers_file: ./speakers.yml
|
|
||||||
autocorrect_file: ./autocorrect.yml
|
|
||||||
glossary_file: ./glossary.yml
|
|
||||||
10
examples/session.s3-audio.yml
Normal file
10
examples/session.s3-audio.yml
Normal file
@@ -0,0 +1,10 @@
|
|||||||
|
session_id: 2026-05-03
|
||||||
|
campaign: sample-campaign
|
||||||
|
date: 2026-05-03
|
||||||
|
title: Sample Session
|
||||||
|
inputs:
|
||||||
|
audio_s3:
|
||||||
|
prefix: audio/
|
||||||
|
speakers_file: ./examples/speakers.yml
|
||||||
|
autocorrect_file: ./examples/autocorrect.yml
|
||||||
|
glossary_file: ./examples/glossary.yml
|
||||||
@@ -1,12 +1,7 @@
|
|||||||
session_id: "{{ session_id }}"
|
session_id: "{{ session_id }}"
|
||||||
campaign: sample-campaign
|
campaign: sample-campaign
|
||||||
date: ""
|
|
||||||
title: ""
|
|
||||||
inputs:
|
inputs:
|
||||||
audio_dir: ./audio
|
audio_dir: ./audio
|
||||||
# Optional S3 input alternative. Do not configure with audio_dir/audio_files.
|
speakers_file: ./examples/speakers.yml
|
||||||
# audio_s3:
|
autocorrect_file: ./examples/autocorrect.yml
|
||||||
# prefix: "audio/{{ session_id }}/"
|
glossary_file: ./examples/glossary.yml
|
||||||
speakers_file: ./speakers.yml
|
|
||||||
autocorrect_file: ./autocorrect.yml
|
|
||||||
glossary_file: ./glossary.yml
|
|
||||||
|
|||||||
2
go.mod
2
go.mod
@@ -4,6 +4,7 @@ go 1.25.0
|
|||||||
|
|
||||||
require (
|
require (
|
||||||
github.com/aws/aws-sdk-go-v2/config v1.32.17
|
github.com/aws/aws-sdk-go-v2/config v1.32.17
|
||||||
|
github.com/aws/aws-sdk-go-v2/credentials v1.19.16
|
||||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0
|
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0
|
||||||
github.com/aws/smithy-go v1.25.1
|
github.com/aws/smithy-go v1.25.1
|
||||||
gopkg.in/yaml.v3 v3.0.1
|
gopkg.in/yaml.v3 v3.0.1
|
||||||
@@ -12,7 +13,6 @@ require (
|
|||||||
require (
|
require (
|
||||||
github.com/aws/aws-sdk-go-v2 v1.41.7 // indirect
|
github.com/aws/aws-sdk-go-v2 v1.41.7 // indirect
|
||||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10 // indirect
|
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10 // indirect
|
||||||
github.com/aws/aws-sdk-go-v2/credentials v1.19.16 // indirect
|
|
||||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.23 // indirect
|
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.23 // indirect
|
||||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23 // indirect
|
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23 // indirect
|
||||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23 // indirect
|
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23 // indirect
|
||||||
|
|||||||
66
internal/app/analyze_artifacts.go
Normal file
66
internal/app/analyze_artifacts.go
Normal file
@@ -0,0 +1,66 @@
|
|||||||
|
package app
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"sort"
|
||||||
|
"strings"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||||
|
)
|
||||||
|
|
||||||
|
type artifactSelectionFlag struct {
|
||||||
|
values []string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *artifactSelectionFlag) String() string {
|
||||||
|
return strings.Join(f.values, ",")
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *artifactSelectionFlag) Set(value string) error {
|
||||||
|
f.values = append(f.values, value)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (f *artifactSelectionFlag) Normalize() ([]string, error) {
|
||||||
|
if len(f.values) == 0 {
|
||||||
|
return nil, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
seen := map[string]struct{}{}
|
||||||
|
out := make([]string, 0, len(f.values))
|
||||||
|
for _, raw := range f.values {
|
||||||
|
for _, part := range strings.Split(raw, ",") {
|
||||||
|
name := strings.TrimSpace(part)
|
||||||
|
if name == "" {
|
||||||
|
return nil, fmt.Errorf("artifact names must be non-empty")
|
||||||
|
}
|
||||||
|
if _, ok := seen[name]; ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
seen[name] = struct{}{}
|
||||||
|
out = append(out, name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
sort.Strings(out)
|
||||||
|
return out, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func validateSelectedAnalyzeArtifacts(cfg *config.Config, selected []string) error {
|
||||||
|
if len(selected) == 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if cfg == nil || cfg.Pipeline == nil || cfg.Pipeline.Scriptorium == nil {
|
||||||
|
return fmt.Errorf("--artifacts requires pipeline.scriptorium.artifacts to be configured")
|
||||||
|
}
|
||||||
|
configured := cfg.Pipeline.Scriptorium.Artifacts
|
||||||
|
if len(configured) == 0 {
|
||||||
|
return fmt.Errorf("--artifacts requires at least one configured artifact in pipeline.scriptorium.artifacts")
|
||||||
|
}
|
||||||
|
for _, name := range selected {
|
||||||
|
if _, ok := configured[name]; !ok {
|
||||||
|
return fmt.Errorf("--artifacts includes unknown artifact %q", name)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
140
internal/app/analyze_artifacts_commands_test.go
Normal file
140
internal/app/analyze_artifacts_commands_test.go
Normal file
@@ -0,0 +1,140 @@
|
|||||||
|
package app
|
||||||
|
|
||||||
|
import (
|
||||||
|
"bytes"
|
||||||
|
"context"
|
||||||
|
"os"
|
||||||
|
"path/filepath"
|
||||||
|
"strings"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestExecuteRunStageArtifactsNonAnalyzeFails(t *testing.T) {
|
||||||
|
workspaceRoot := t.TempDir()
|
||||||
|
pipelinePath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
|
||||||
|
|
||||||
|
var stdout bytes.Buffer
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
code := Execute(
|
||||||
|
[]string{"run-stage", "--config", pipelinePath, "--session", sessionPath, "--artifacts", "session_recap", "polish"},
|
||||||
|
&stdout,
|
||||||
|
&stderr,
|
||||||
|
)
|
||||||
|
if code == 0 {
|
||||||
|
t.Fatal("exit code = 0, want non-zero")
|
||||||
|
}
|
||||||
|
if !strings.Contains(stderr.String(), `run-stage: --artifacts is only supported for stage "analyze"`) {
|
||||||
|
t.Fatalf("stderr = %q, want stage-gating error", stderr.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestExecuteUnknownArtifactsFailValidation(t *testing.T) {
|
||||||
|
workspaceRoot := t.TempDir()
|
||||||
|
pipelinePath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
|
||||||
|
|
||||||
|
var stdout bytes.Buffer
|
||||||
|
var stderr bytes.Buffer
|
||||||
|
code := Execute(
|
||||||
|
[]string{"run", "--config", pipelinePath, "--session", sessionPath, "--artifacts", "unknown_artifact"},
|
||||||
|
&stdout,
|
||||||
|
&stderr,
|
||||||
|
)
|
||||||
|
if code == 0 {
|
||||||
|
t.Fatal("exit code = 0, want non-zero")
|
||||||
|
}
|
||||||
|
if !strings.Contains(stderr.String(), `run: --artifacts includes unknown artifact "unknown_artifact"`) {
|
||||||
|
t.Fatalf("stderr = %q, want unknown-artifact validation error", stderr.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestRunStageArtifactsDoesNotImplyForce(t *testing.T) {
|
||||||
|
workspaceRoot := t.TempDir()
|
||||||
|
pipelinePath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
|
||||||
|
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
|
||||||
|
|
||||||
|
store := &manifest.LocalStore{}
|
||||||
|
seed := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
|
||||||
|
seed.MarkStageSucceeded("analyze", time.Date(2026, 5, 3, 10, 1, 0, 0, time.UTC), nil)
|
||||||
|
if err := store.Save(context.Background(), manifestPath, seed); err != nil {
|
||||||
|
t.Fatalf("save manifest: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
var out bytes.Buffer
|
||||||
|
err := RunStage(
|
||||||
|
context.Background(),
|
||||||
|
[]string{"--config", pipelinePath, "--session", sessionPath, "--artifacts", "session_recap,session_recap", "analyze"},
|
||||||
|
&out,
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("RunStage() error = %v", err)
|
||||||
|
}
|
||||||
|
if !strings.Contains(out.String(), "stage=analyze executed=0 skipped=1 force=false") {
|
||||||
|
t.Fatalf("output = %q, want analyze skip without force", out.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestResumeArtifactsWithSucceededAnalyzeSkipsUnlessForced(t *testing.T) {
|
||||||
|
workspaceRoot := t.TempDir()
|
||||||
|
pipelinePath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
|
||||||
|
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
|
||||||
|
|
||||||
|
store := &manifest.LocalStore{}
|
||||||
|
seed := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
|
||||||
|
for _, stageName := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "analyze", "archive", "notify"} {
|
||||||
|
seed.MarkStageSucceeded(stageName, time.Date(2026, 5, 3, 10, 1, 0, 0, time.UTC), nil)
|
||||||
|
}
|
||||||
|
if err := store.Save(context.Background(), manifestPath, seed); err != nil {
|
||||||
|
t.Fatalf("save manifest: %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
var out bytes.Buffer
|
||||||
|
err := Resume(
|
||||||
|
context.Background(),
|
||||||
|
[]string{"--config", pipelinePath, "--session", sessionPath, "--artifacts", "session_recap"},
|
||||||
|
&out,
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Resume() error = %v", err)
|
||||||
|
}
|
||||||
|
if !strings.Contains(out.String(), "has no remaining stages") {
|
||||||
|
t.Fatalf("output = %q, want no remaining stages", out.String())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func writeValidConfigFilesWithScriptoriumArtifacts(t *testing.T, workspaceRoot string) (string, string) {
|
||||||
|
t.Helper()
|
||||||
|
|
||||||
|
pipelinePath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
|
||||||
|
f, err := os.OpenFile(pipelinePath, os.O_APPEND|os.O_WRONLY, 0)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("open pipeline config for append: %v", err)
|
||||||
|
}
|
||||||
|
defer f.Close()
|
||||||
|
|
||||||
|
extra := `
|
||||||
|
scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
artifacts:
|
||||||
|
session_recap:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd.session_recap
|
||||||
|
output_path: artifacts/session_recap.md
|
||||||
|
player_handout:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd.player_handout
|
||||||
|
output_path: artifacts/player_handout.md
|
||||||
|
depends_on:
|
||||||
|
- session_recap
|
||||||
|
inputs:
|
||||||
|
recap:
|
||||||
|
source: narratio.artifact.session_recap
|
||||||
|
required: true
|
||||||
|
`
|
||||||
|
if _, err := f.WriteString(extra); err != nil {
|
||||||
|
t.Fatalf("append scriptorium config: %v", err)
|
||||||
|
}
|
||||||
|
return pipelinePath, sessionPath
|
||||||
|
}
|
||||||
132
internal/app/analyze_artifacts_test.go
Normal file
132
internal/app/analyze_artifacts_test.go
Normal file
@@ -0,0 +1,132 @@
|
|||||||
|
package app
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestArtifactSelectionFlagNormalize(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
inputs []string
|
||||||
|
want []string
|
||||||
|
wantErr string
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "single value",
|
||||||
|
inputs: []string{"session_recap"},
|
||||||
|
want: []string{"session_recap"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "repeatable and comma separated values are deduped and sorted",
|
||||||
|
inputs: []string{"session_recap,player_handout", "session_recap"},
|
||||||
|
want: []string{"player_handout", "session_recap"},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "empty token fails",
|
||||||
|
inputs: []string{"session_recap,"},
|
||||||
|
wantErr: "artifact names must be non-empty",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
var flag artifactSelectionFlag
|
||||||
|
for _, in := range tt.inputs {
|
||||||
|
if err := flag.Set(in); err != nil {
|
||||||
|
t.Fatalf("Set(%q) error = %v", in, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
got, err := flag.Normalize()
|
||||||
|
if tt.wantErr != "" {
|
||||||
|
if err == nil {
|
||||||
|
t.Fatalf("Normalize() error = nil, want %q", tt.wantErr)
|
||||||
|
}
|
||||||
|
if err.Error() != tt.wantErr {
|
||||||
|
t.Fatalf("Normalize() error = %q, want %q", err.Error(), tt.wantErr)
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Normalize() error = %v", err)
|
||||||
|
}
|
||||||
|
if len(got) != len(tt.want) {
|
||||||
|
t.Fatalf("Normalize() len = %d, want %d; got=%v", len(got), len(tt.want), got)
|
||||||
|
}
|
||||||
|
for i := range got {
|
||||||
|
if got[i] != tt.want[i] {
|
||||||
|
t.Fatalf("Normalize()[%d] = %q, want %q", i, got[i], tt.want[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestValidateSelectedAnalyzeArtifacts(t *testing.T) {
|
||||||
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
cfg *config.Config
|
||||||
|
selected []string
|
||||||
|
wantErr string
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "empty selection is accepted",
|
||||||
|
cfg: &config.Config{},
|
||||||
|
selected: nil,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "scriptorium required when selected artifacts present",
|
||||||
|
cfg: &config.Config{Pipeline: &config.PipelineConfig{}},
|
||||||
|
selected: []string{"session_recap"},
|
||||||
|
wantErr: "--artifacts requires pipeline.scriptorium.artifacts to be configured",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "unknown selected artifact fails",
|
||||||
|
cfg: &config.Config{
|
||||||
|
Pipeline: &config.PipelineConfig{
|
||||||
|
Scriptorium: &config.ScriptoriumConfig{
|
||||||
|
Artifacts: map[string]config.ScriptoriumArtifactConfig{
|
||||||
|
"session_recap": {Enabled: true, PromptID: "dnd.session_recap", OutputPath: "artifacts/session_recap.md"},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
selected: []string{"player_handout"},
|
||||||
|
wantErr: `--artifacts includes unknown artifact "player_handout"`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "known selected artifacts are accepted",
|
||||||
|
cfg: &config.Config{
|
||||||
|
Pipeline: &config.PipelineConfig{
|
||||||
|
Scriptorium: &config.ScriptoriumConfig{
|
||||||
|
Artifacts: map[string]config.ScriptoriumArtifactConfig{
|
||||||
|
"session_recap": {Enabled: true, PromptID: "dnd.session_recap", OutputPath: "artifacts/session_recap.md"},
|
||||||
|
"player_handout": {Enabled: true, PromptID: "dnd.player_handout", OutputPath: "artifacts/player_handout.md"},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
selected: []string{"player_handout", "session_recap"},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
err := validateSelectedAnalyzeArtifacts(tt.cfg, tt.selected)
|
||||||
|
if tt.wantErr != "" {
|
||||||
|
if err == nil {
|
||||||
|
t.Fatalf("error = nil, want %q", tt.wantErr)
|
||||||
|
}
|
||||||
|
if err.Error() != tt.wantErr {
|
||||||
|
t.Fatalf("error = %q, want %q", err.Error(), tt.wantErr)
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("error = %v, want nil", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -20,10 +20,12 @@ func Resume(ctx context.Context, args []string, out io.Writer) error {
|
|||||||
var sessionPath string
|
var sessionPath string
|
||||||
var sessionID string
|
var sessionID string
|
||||||
var force bool
|
var force bool
|
||||||
|
var selectedArtifacts artifactSelectionFlag
|
||||||
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
|
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
|
||||||
fs.StringVar(&sessionPath, "session", "", "path to session.yml")
|
fs.StringVar(&sessionPath, "session", "", "path to session.yml")
|
||||||
fs.StringVar(&sessionID, "session-id", "", "session identifier for session.yml templates")
|
fs.StringVar(&sessionID, "session-id", "", "session identifier for session.yml templates")
|
||||||
fs.BoolVar(&force, "force", false, "force stage execution")
|
fs.BoolVar(&force, "force", false, "force stage execution")
|
||||||
|
fs.Var(&selectedArtifacts, "artifacts", "artifact names to execute during analyze (comma-separated or repeatable)")
|
||||||
|
|
||||||
if err := fs.Parse(args); err != nil {
|
if err := fs.Parse(args); err != nil {
|
||||||
return fmt.Errorf("resume: invalid flags: %w", err)
|
return fmt.Errorf("resume: invalid flags: %w", err)
|
||||||
@@ -49,6 +51,13 @@ func Resume(ctx context.Context, args []string, out io.Writer) error {
|
|||||||
if err := config.Validate(cfg); err != nil {
|
if err := config.Validate(cfg); err != nil {
|
||||||
return fmt.Errorf("resume: %w", err)
|
return fmt.Errorf("resume: %w", err)
|
||||||
}
|
}
|
||||||
|
normalizedArtifacts, err := selectedArtifacts.Normalize()
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("resume: invalid --artifacts: %w", err)
|
||||||
|
}
|
||||||
|
if err := validateSelectedAnalyzeArtifacts(cfg, normalizedArtifacts); err != nil {
|
||||||
|
return fmt.Errorf("resume: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
full := BuildFullPlan()
|
full := BuildFullPlan()
|
||||||
selected := full
|
selected := full
|
||||||
@@ -67,7 +76,10 @@ func Resume(ctx context.Context, args []string, out io.Writer) error {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
summary, err := executeStages(ctx, cfg, selected, RunOptions{Force: force})
|
summary, err := executeStages(ctx, cfg, selected, RunOptions{
|
||||||
|
Force: force,
|
||||||
|
SelectedArtifacts: normalizedArtifacts,
|
||||||
|
})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("resume: %w", err)
|
return fmt.Errorf("resume: %w", err)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -18,10 +18,12 @@ func Run(ctx context.Context, args []string, out io.Writer) error {
|
|||||||
var sessionPath string
|
var sessionPath string
|
||||||
var sessionID string
|
var sessionID string
|
||||||
var force bool
|
var force bool
|
||||||
|
var selectedArtifacts artifactSelectionFlag
|
||||||
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
|
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
|
||||||
fs.StringVar(&sessionPath, "session", "", "path to session.yml")
|
fs.StringVar(&sessionPath, "session", "", "path to session.yml")
|
||||||
fs.StringVar(&sessionID, "session-id", "", "session identifier for session.yml templates")
|
fs.StringVar(&sessionID, "session-id", "", "session identifier for session.yml templates")
|
||||||
fs.BoolVar(&force, "force", false, "force stage execution (reserved for future behavior)")
|
fs.BoolVar(&force, "force", false, "force stage execution (reserved for future behavior)")
|
||||||
|
fs.Var(&selectedArtifacts, "artifacts", "artifact names to execute during analyze (comma-separated or repeatable)")
|
||||||
|
|
||||||
if err := fs.Parse(args); err != nil {
|
if err := fs.Parse(args); err != nil {
|
||||||
return fmt.Errorf("run: invalid flags: %w", err)
|
return fmt.Errorf("run: invalid flags: %w", err)
|
||||||
@@ -47,9 +49,19 @@ func Run(ctx context.Context, args []string, out io.Writer) error {
|
|||||||
if err := config.Validate(cfg); err != nil {
|
if err := config.Validate(cfg); err != nil {
|
||||||
return fmt.Errorf("run: %w", err)
|
return fmt.Errorf("run: %w", err)
|
||||||
}
|
}
|
||||||
|
normalizedArtifacts, err := selectedArtifacts.Normalize()
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("run: invalid --artifacts: %w", err)
|
||||||
|
}
|
||||||
|
if err := validateSelectedAnalyzeArtifacts(cfg, normalizedArtifacts); err != nil {
|
||||||
|
return fmt.Errorf("run: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
stages := BuildFullPlan()
|
stages := BuildFullPlan()
|
||||||
summary, err := executeStages(ctx, cfg, stages, RunOptions{Force: force})
|
summary, err := executeStages(ctx, cfg, stages, RunOptions{
|
||||||
|
Force: force,
|
||||||
|
SelectedArtifacts: normalizedArtifacts,
|
||||||
|
})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("run: %w", err)
|
return fmt.Errorf("run: %w", err)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -18,10 +18,12 @@ func RunStage(ctx context.Context, args []string, out io.Writer) error {
|
|||||||
var sessionPath string
|
var sessionPath string
|
||||||
var sessionID string
|
var sessionID string
|
||||||
var force bool
|
var force bool
|
||||||
|
var selectedArtifacts artifactSelectionFlag
|
||||||
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
|
fs.StringVar(&pipelinePath, "config", "", "path to pipeline.yml (optional; defaults searched)")
|
||||||
fs.StringVar(&sessionPath, "session", "", "path to session.yml")
|
fs.StringVar(&sessionPath, "session", "", "path to session.yml")
|
||||||
fs.StringVar(&sessionID, "session-id", "", "session identifier for session.yml templates")
|
fs.StringVar(&sessionID, "session-id", "", "session identifier for session.yml templates")
|
||||||
fs.BoolVar(&force, "force", false, "force stage execution (reserved for future behavior)")
|
fs.BoolVar(&force, "force", false, "force stage execution (reserved for future behavior)")
|
||||||
|
fs.Var(&selectedArtifacts, "artifacts", "artifact names to execute during analyze (comma-separated or repeatable)")
|
||||||
|
|
||||||
if err := fs.Parse(args); err != nil {
|
if err := fs.Parse(args); err != nil {
|
||||||
return fmt.Errorf("run-stage: invalid flags: %w", err)
|
return fmt.Errorf("run-stage: invalid flags: %w", err)
|
||||||
@@ -30,6 +32,13 @@ func RunStage(ctx context.Context, args []string, out io.Writer) error {
|
|||||||
return fmt.Errorf("run-stage: expected exactly one stage name")
|
return fmt.Errorf("run-stage: expected exactly one stage name")
|
||||||
}
|
}
|
||||||
stageName := fs.Arg(0)
|
stageName := fs.Arg(0)
|
||||||
|
normalizedArtifacts, err := selectedArtifacts.Normalize()
|
||||||
|
if err != nil {
|
||||||
|
return fmt.Errorf("run-stage: invalid --artifacts: %w", err)
|
||||||
|
}
|
||||||
|
if len(normalizedArtifacts) > 0 && stageName != "analyze" {
|
||||||
|
return fmt.Errorf("run-stage: --artifacts is only supported for stage \"analyze\"")
|
||||||
|
}
|
||||||
stages, err := BuildSingleStagePlan(stageName)
|
stages, err := BuildSingleStagePlan(stageName)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("run-stage: %w", err)
|
return fmt.Errorf("run-stage: %w", err)
|
||||||
@@ -53,8 +62,14 @@ func RunStage(ctx context.Context, args []string, out io.Writer) error {
|
|||||||
if err := config.Validate(cfg); err != nil {
|
if err := config.Validate(cfg); err != nil {
|
||||||
return fmt.Errorf("run-stage: %w", err)
|
return fmt.Errorf("run-stage: %w", err)
|
||||||
}
|
}
|
||||||
|
if err := validateSelectedAnalyzeArtifacts(cfg, normalizedArtifacts); err != nil {
|
||||||
|
return fmt.Errorf("run-stage: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
summary, err := executeStages(ctx, cfg, stages, RunOptions{Force: force})
|
summary, err := executeStages(ctx, cfg, stages, RunOptions{
|
||||||
|
Force: force,
|
||||||
|
SelectedArtifacts: normalizedArtifacts,
|
||||||
|
})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return fmt.Errorf("run-stage: %w", err)
|
return fmt.Errorf("run-stage: %w", err)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -22,6 +22,7 @@ import (
|
|||||||
|
|
||||||
type RunOptions struct {
|
type RunOptions struct {
|
||||||
Force bool
|
Force bool
|
||||||
|
SelectedArtifacts []string
|
||||||
Env *Env
|
Env *Env
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -43,6 +44,7 @@ func executeStages(ctx context.Context, cfg *config.Config, stages []stage.Stage
|
|||||||
if env.Config == nil {
|
if env.Config == nil {
|
||||||
env.Config = cfg
|
env.Config = cfg
|
||||||
}
|
}
|
||||||
|
env.SelectedAnalyzeArtifacts = append([]string(nil), opts.SelectedArtifacts...)
|
||||||
if env.ArtifactStore == nil {
|
if env.ArtifactStore == nil {
|
||||||
env.ArtifactStore = artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root)
|
env.ArtifactStore = artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root)
|
||||||
}
|
}
|
||||||
@@ -202,7 +204,7 @@ func executeStages(ctx context.Context, cfg *config.Config, stages []stage.Stage
|
|||||||
return nil, fmt.Errorf("stage %q failed: %w", s.Name(), err)
|
return nil, fmt.Errorf("stage %q failed: %w", s.Name(), err)
|
||||||
}
|
}
|
||||||
|
|
||||||
outputs := mapResultOutputs(result, runID)
|
outputs := mapResultOutputs(s.Name(), result, runID)
|
||||||
succeededAt := nowUTC()
|
succeededAt := nowUTC()
|
||||||
m.MarkStageSucceeded(s.Name(), succeededAt, outputs)
|
m.MarkStageSucceeded(s.Name(), succeededAt, outputs)
|
||||||
applyStageResultToManifest(m, s.Name(), result)
|
applyStageResultToManifest(m, s.Name(), result)
|
||||||
@@ -386,7 +388,7 @@ func fileExists(path string) (bool, error) {
|
|||||||
return false, err
|
return false, err
|
||||||
}
|
}
|
||||||
|
|
||||||
func mapResultOutputs(result *stage.StageResult, runID string) []manifest.ArtifactRecord {
|
func mapResultOutputs(stageName string, result *stage.StageResult, runID string) []manifest.ArtifactRecord {
|
||||||
if result == nil || len(result.Outputs) == 0 {
|
if result == nil || len(result.Outputs) == 0 {
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
@@ -398,8 +400,15 @@ func mapResultOutputs(result *stage.StageResult, runID string) []manifest.Artifa
|
|||||||
if localPath == "" {
|
if localPath == "" {
|
||||||
localPath = ref.RelativePath
|
localPath = ref.RelativePath
|
||||||
}
|
}
|
||||||
|
kind := ref.Kind
|
||||||
|
sourceID := ""
|
||||||
|
if stageName == "analyze" {
|
||||||
|
sourceID = artifacts.ConfiguredArtifactSourceID(ref.Kind)
|
||||||
|
kind = "scriptorium_artifact"
|
||||||
|
}
|
||||||
out = append(out, manifest.ArtifactRecord{
|
out = append(out, manifest.ArtifactRecord{
|
||||||
Kind: ref.Kind,
|
Kind: kind,
|
||||||
|
SourceID: sourceID,
|
||||||
LocalPath: localPath,
|
LocalPath: localPath,
|
||||||
ProducerRunID: runID,
|
ProducerRunID: runID,
|
||||||
RemoteKey: ref.RemoteKey,
|
RemoteKey: ref.RemoteKey,
|
||||||
|
|||||||
@@ -3,6 +3,7 @@ package app
|
|||||||
import (
|
import (
|
||||||
"context"
|
"context"
|
||||||
"errors"
|
"errors"
|
||||||
|
"fmt"
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
"strings"
|
"strings"
|
||||||
@@ -44,6 +45,213 @@ func (s countingStage) Run(_ context.Context, _ *stage.Env, _ *manifest.Manifest
|
|||||||
return &stage.StageResult{Metadata: map[string]any{"counting": true}}, nil
|
return &stage.StageResult{Metadata: map[string]any{"counting": true}}, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
type captureSelectedArtifactsStage struct {
|
||||||
|
name string
|
||||||
|
captured *[]string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s captureSelectedArtifactsStage) Name() string { return s.name }
|
||||||
|
func (s captureSelectedArtifactsStage) Declares() stage.IODecl { return stage.IODecl{} }
|
||||||
|
func (s captureSelectedArtifactsStage) Run(_ context.Context, env *stage.Env, _ *manifest.Manifest) (*stage.StageResult, error) {
|
||||||
|
if s.captured != nil {
|
||||||
|
*s.captured = append((*s.captured)[:0], env.SelectedAnalyzeArtifacts...)
|
||||||
|
}
|
||||||
|
return &stage.StageResult{Metadata: map[string]any{"captured": true}}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type analyzeOutputStage struct {
|
||||||
|
output artifacts.Ref
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s analyzeOutputStage) Name() string { return "analyze" }
|
||||||
|
func (s analyzeOutputStage) Declares() stage.IODecl { return stage.IODecl{} }
|
||||||
|
func (s analyzeOutputStage) Run(_ context.Context, _ *stage.Env, _ *manifest.Manifest) (*stage.StageResult, error) {
|
||||||
|
return &stage.StageResult{
|
||||||
|
Outputs: []artifacts.Ref{s.output},
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
type selectedAnalyzeArtifactStage struct {
|
||||||
|
expected []string
|
||||||
|
}
|
||||||
|
|
||||||
|
func (s selectedAnalyzeArtifactStage) Name() string { return "analyze" }
|
||||||
|
func (s selectedAnalyzeArtifactStage) Declares() stage.IODecl { return stage.IODecl{} }
|
||||||
|
func (s selectedAnalyzeArtifactStage) Run(_ context.Context, env *stage.Env, m *manifest.Manifest) (*stage.StageResult, error) {
|
||||||
|
if len(env.SelectedAnalyzeArtifacts) != len(s.expected) {
|
||||||
|
return nil, fmt.Errorf("selected artifacts len = %d, want %d", len(env.SelectedAnalyzeArtifacts), len(s.expected))
|
||||||
|
}
|
||||||
|
for i := range s.expected {
|
||||||
|
if env.SelectedAnalyzeArtifacts[i] != s.expected[i] {
|
||||||
|
return nil, fmt.Errorf("selected artifacts[%d] = %q, want %q", i, env.SelectedAnalyzeArtifacts[i], s.expected[i])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
outputPath := filepath.Join(
|
||||||
|
artifacts.SessionWorkDirForCampaign(env.Config.Pipeline.Workspace.Root, env.Config.Session.Campaign, m.SessionID),
|
||||||
|
"artifacts",
|
||||||
|
"player_handout.md",
|
||||||
|
)
|
||||||
|
if err := os.MkdirAll(filepath.Dir(outputPath), 0o755); err != nil {
|
||||||
|
return nil, fmt.Errorf("mkdir artifact dir: %w", err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(outputPath, []byte("player handout\n"), 0o644); err != nil {
|
||||||
|
return nil, fmt.Errorf("write player handout: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
return &stage.StageResult{
|
||||||
|
Outputs: []artifacts.Ref{
|
||||||
|
{
|
||||||
|
Kind: "player_handout",
|
||||||
|
Category: "artifacts",
|
||||||
|
RelativePath: "artifacts/player_handout.md",
|
||||||
|
AbsolutePath: outputPath,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
Metadata: map[string]any{
|
||||||
|
"stage": "analyze",
|
||||||
|
},
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestExecuteStagesPropagatesSelectedArtifactsToEnv(t *testing.T) {
|
||||||
|
cfg := testConfig(t)
|
||||||
|
|
||||||
|
captured := []string{}
|
||||||
|
stageToRun := captureSelectedArtifactsStage{name: "analyze", captured: &captured}
|
||||||
|
_, err := executeStages(context.Background(), cfg, []stage.Stage{stageToRun}, RunOptions{
|
||||||
|
SelectedArtifacts: []string{"player_handout", "session_recap"},
|
||||||
|
})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("executeStages() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(captured) != 2 {
|
||||||
|
t.Fatalf("captured len = %d, want 2 (%v)", len(captured), captured)
|
||||||
|
}
|
||||||
|
if captured[0] != "player_handout" || captured[1] != "session_recap" {
|
||||||
|
t.Fatalf("captured = %v, want [player_handout session_recap]", captured)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestExecuteStagesAnalyzeOutputsPersistAsScriptoriumArtifacts(t *testing.T) {
|
||||||
|
cfg := testConfig(t)
|
||||||
|
storeForPaths := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root)
|
||||||
|
sessionPaths := storeForPaths.SessionPathsFor(cfg.Session.Campaign, cfg.Session.SessionID)
|
||||||
|
outputPath := filepath.Join(sessionPaths.ArtifactsDir, "session_recap.md")
|
||||||
|
|
||||||
|
stageToRun := analyzeOutputStage{
|
||||||
|
output: artifacts.Ref{
|
||||||
|
Kind: "session_recap",
|
||||||
|
Category: "artifacts",
|
||||||
|
RelativePath: "artifacts/session_recap.md",
|
||||||
|
AbsolutePath: outputPath,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
summary, err := executeStages(context.Background(), cfg, []stage.Stage{stageToRun}, RunOptions{})
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("executeStages() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
store := &manifest.LocalStore{}
|
||||||
|
sessionManifest, err := store.Load(context.Background(), summary.ManifestPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("load session manifest: %v", err)
|
||||||
|
}
|
||||||
|
sessionStage := sessionManifest.Stages["analyze"]
|
||||||
|
if sessionStage == nil {
|
||||||
|
t.Fatal("session manifest analyze stage missing")
|
||||||
|
}
|
||||||
|
if len(sessionStage.Outputs) != 1 {
|
||||||
|
t.Fatalf("session analyze outputs len = %d, want 1", len(sessionStage.Outputs))
|
||||||
|
}
|
||||||
|
sessionOutput := sessionStage.Outputs[0]
|
||||||
|
if sessionOutput.Kind != "scriptorium_artifact" {
|
||||||
|
t.Fatalf("session output kind = %q, want scriptorium_artifact", sessionOutput.Kind)
|
||||||
|
}
|
||||||
|
if sessionOutput.SourceID != "narratio.artifact.session_recap" {
|
||||||
|
t.Fatalf("session output source_id = %q, want narratio.artifact.session_recap", sessionOutput.SourceID)
|
||||||
|
}
|
||||||
|
if sessionOutput.LocalPath != outputPath {
|
||||||
|
t.Fatalf("session output local_path = %q, want %q", sessionOutput.LocalPath, outputPath)
|
||||||
|
}
|
||||||
|
|
||||||
|
runManifest, err := store.LoadRun(context.Background(), summary.RunManifestPath)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("load run manifest: %v", err)
|
||||||
|
}
|
||||||
|
runStage := runManifest.Stages["analyze"]
|
||||||
|
if runStage == nil {
|
||||||
|
t.Fatal("run manifest analyze stage missing")
|
||||||
|
}
|
||||||
|
if len(runStage.Outputs) != 1 {
|
||||||
|
t.Fatalf("run analyze outputs len = %d, want 1", len(runStage.Outputs))
|
||||||
|
}
|
||||||
|
runOutput := runStage.Outputs[0]
|
||||||
|
if runOutput.Kind != "scriptorium_artifact" {
|
||||||
|
t.Fatalf("run output kind = %q, want scriptorium_artifact", runOutput.Kind)
|
||||||
|
}
|
||||||
|
if runOutput.SourceID != "narratio.artifact.session_recap" {
|
||||||
|
t.Fatalf("run output source_id = %q, want narratio.artifact.session_recap", runOutput.SourceID)
|
||||||
|
}
|
||||||
|
if runOutput.LocalPath != outputPath {
|
||||||
|
t.Fatalf("run output local_path = %q, want %q", runOutput.LocalPath, outputPath)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestExecuteStagesArchiveFailsWhenRequiredRecapPromotionMissingForSelectedArtifacts(t *testing.T) {
|
||||||
|
cfg := testConfig(t)
|
||||||
|
cfg.Pipeline.Storage.S3 = &config.StorageS3Config{
|
||||||
|
Bucket: "my-dnd-archive",
|
||||||
|
RootPrefix: "dnd",
|
||||||
|
}
|
||||||
|
cfg.Pipeline.Archive = &config.ArchiveConfig{
|
||||||
|
Enabled: boolPtr(true),
|
||||||
|
UploadRun: boolPtr(true),
|
||||||
|
PromoteArtifacts: []config.ArchivePromotionRule{
|
||||||
|
{From: "artifacts/session_recap.md", To: "artifacts/session_recap.md", Required: boolPtr(true)},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
store := &manifest.LocalStore{}
|
||||||
|
manifestPath := manifestPathFor(cfg)
|
||||||
|
seed := manifest.New(cfg.Session.SessionID, time.Now().UTC())
|
||||||
|
seed.Campaign = cfg.Session.Campaign
|
||||||
|
for _, stageName := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim"} {
|
||||||
|
seed.MarkStageSucceeded(stageName, time.Now().UTC(), nil)
|
||||||
|
}
|
||||||
|
if err := os.MkdirAll(filepath.Dir(manifestPath), 0o755); err != nil {
|
||||||
|
t.Fatalf("MkdirAll() error = %v", err)
|
||||||
|
}
|
||||||
|
if err := store.Save(context.Background(), manifestPath, seed); err != nil {
|
||||||
|
t.Fatalf("Save manifest error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
archiveStageImpl, err := stage.Select("archive")
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Select(archive) error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
_, err = executeStages(
|
||||||
|
context.Background(),
|
||||||
|
cfg,
|
||||||
|
[]stage.Stage{
|
||||||
|
selectedAnalyzeArtifactStage{expected: []string{"player_handout"}},
|
||||||
|
archiveStageImpl,
|
||||||
|
},
|
||||||
|
RunOptions{
|
||||||
|
SelectedArtifacts: []string{"player_handout"},
|
||||||
|
Env: &Env{ObjectStore: &storage.FakeBackend{}},
|
||||||
|
},
|
||||||
|
)
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected archive promotion failure, got nil")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), "required promotion source missing") {
|
||||||
|
t.Fatalf("error = %q, want required promotion source missing", err.Error())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestExecuteStagesPlaceholderSuccessUpdatesManifest(t *testing.T) {
|
func TestExecuteStagesPlaceholderSuccessUpdatesManifest(t *testing.T) {
|
||||||
cfg := testConfig(t)
|
cfg := testConfig(t)
|
||||||
|
|
||||||
@@ -662,7 +870,7 @@ func TestAdapterBackedStageFailureMarksManifestFailed(t *testing.T) {
|
|||||||
PromptID: "dnd.session_recap",
|
PromptID: "dnd.session_recap",
|
||||||
OutputPath: "artifacts/session_recap.md",
|
OutputPath: "artifacts/session_recap.md",
|
||||||
Inputs: map[string]config.ScriptoriumInputConfig{
|
Inputs: map[string]config.ScriptoriumInputConfig{
|
||||||
"transcript": {Source: "processed_transcript", Required: true},
|
"transcript": {Source: "narratio.transcript.polished", Required: true},
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -6,6 +6,7 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"os"
|
"os"
|
||||||
"path/filepath"
|
"path/filepath"
|
||||||
|
"regexp"
|
||||||
"strings"
|
"strings"
|
||||||
|
|
||||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||||
@@ -17,11 +18,11 @@ const (
|
|||||||
ArtifactTranscriptFull = "narratio.transcript.full"
|
ArtifactTranscriptFull = "narratio.transcript.full"
|
||||||
ArtifactTranscriptTrimmed = "narratio.transcript.trimmed"
|
ArtifactTranscriptTrimmed = "narratio.transcript.trimmed"
|
||||||
ArtifactBoundsSession = "narratio.bounds.session"
|
ArtifactBoundsSession = "narratio.bounds.session"
|
||||||
ArtifactSessionRecap = "narratio.artifact.session_recap"
|
|
||||||
)
|
)
|
||||||
|
|
||||||
// ErrSessionArtifactNotFound is returned when no readable artifact exists for a known ID.
|
// ErrSessionArtifactNotFound is returned when no readable artifact exists for a known ID.
|
||||||
var ErrSessionArtifactNotFound = errors.New("session artifact not found")
|
var ErrSessionArtifactNotFound = errors.New("session artifact not found")
|
||||||
|
var configuredArtifactSourceRE = regexp.MustCompile(`^narratio\.artifact\.[a-z][a-z0-9_]*$`)
|
||||||
|
|
||||||
type artifactContentKind string
|
type artifactContentKind string
|
||||||
|
|
||||||
@@ -75,19 +76,6 @@ var artifactRegistry = map[string]artifactSpec{
|
|||||||
OutputKind: "session_bounds",
|
OutputKind: "session_bounds",
|
||||||
ContentKind: contentJSON,
|
ContentKind: contentJSON,
|
||||||
},
|
},
|
||||||
ArtifactSessionRecap: {
|
|
||||||
ID: ArtifactSessionRecap,
|
|
||||||
CanonicalRelPath: "artifacts/session_recap.md",
|
|
||||||
ProducerStage: "analyze",
|
|
||||||
OutputKind: "session_recap",
|
|
||||||
ContentKind: contentText,
|
|
||||||
},
|
|
||||||
}
|
|
||||||
|
|
||||||
var artifactAliases = map[string]string{
|
|
||||||
"processed_transcript": ArtifactTranscriptPolished,
|
|
||||||
"normalized_transcript": ArtifactTranscriptFull,
|
|
||||||
"trimmed_transcript": ArtifactTranscriptTrimmed,
|
|
||||||
}
|
}
|
||||||
|
|
||||||
// ResolvedSessionArtifact describes one session-level artifact lookup result.
|
// ResolvedSessionArtifact describes one session-level artifact lookup result.
|
||||||
@@ -113,21 +101,23 @@ func (e *SessionArtifactNotFoundError) Unwrap() error {
|
|||||||
return ErrSessionArtifactNotFound
|
return ErrSessionArtifactNotFound
|
||||||
}
|
}
|
||||||
|
|
||||||
// NormalizeSessionArtifactSource maps legacy aliases to canonical IDs and validates IDs.
|
// NormalizeSessionArtifactSource validates canonical artifact IDs.
|
||||||
func NormalizeSessionArtifactSource(source string) (string, error) {
|
func NormalizeSessionArtifactSource(source string) (string, error) {
|
||||||
normalized := strings.TrimSpace(source)
|
normalized := strings.TrimSpace(source)
|
||||||
if normalized == "" {
|
if normalized == "" {
|
||||||
return "", fmt.Errorf("artifact source is required")
|
return "", fmt.Errorf("artifact source is required")
|
||||||
}
|
}
|
||||||
if alias, ok := artifactAliases[normalized]; ok {
|
|
||||||
normalized = alias
|
|
||||||
}
|
|
||||||
if _, ok := artifactRegistry[normalized]; !ok {
|
if _, ok := artifactRegistry[normalized]; !ok {
|
||||||
return "", fmt.Errorf("unsupported artifact source %q", source)
|
return "", fmt.Errorf("unsupported artifact source %q", source)
|
||||||
}
|
}
|
||||||
return normalized, nil
|
return normalized, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// IsConfiguredArtifactSource returns true when source is narratio.artifact.<name>.
|
||||||
|
func IsConfiguredArtifactSource(source string) bool {
|
||||||
|
return configuredArtifactSourceRE.MatchString(strings.TrimSpace(source))
|
||||||
|
}
|
||||||
|
|
||||||
// ResolveSessionArtifact resolves a symbolic source to a readable local session artifact path.
|
// ResolveSessionArtifact resolves a symbolic source to a readable local session artifact path.
|
||||||
// Resolution order is manifest producer outputs first, then canonical session path fallback.
|
// Resolution order is manifest producer outputs first, then canonical session path fallback.
|
||||||
func ResolveSessionArtifact(paths SessionPaths, m *manifest.Manifest, source string) (ResolvedSessionArtifact, error) {
|
func ResolveSessionArtifact(paths SessionPaths, m *manifest.Manifest, source string) (ResolvedSessionArtifact, error) {
|
||||||
@@ -176,6 +166,35 @@ func ResolveSessionArtifact(paths SessionPaths, m *manifest.Manifest, source str
|
|||||||
return ResolvedSessionArtifact{}, &SessionArtifactNotFoundError{ArtifactID: spec.ID}
|
return ResolvedSessionArtifact{}, &SessionArtifactNotFoundError{ArtifactID: spec.ID}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// ResolveSessionArtifactWithCatalog resolves built-in sources using existing rules and resolves
|
||||||
|
// configured narratio.artifact.<name> sources through runtime catalog availability.
|
||||||
|
func ResolveSessionArtifactWithCatalog(paths SessionPaths, m *manifest.Manifest, source string, catalog *ArtifactCatalog) (ResolvedSessionArtifact, error) {
|
||||||
|
normalized := strings.TrimSpace(source)
|
||||||
|
if !IsConfiguredArtifactSource(normalized) {
|
||||||
|
return ResolveSessionArtifact(paths, m, normalized)
|
||||||
|
}
|
||||||
|
if catalog == nil {
|
||||||
|
return ResolvedSessionArtifact{}, fmt.Errorf("configured artifact source %q requires runtime artifact catalog", source)
|
||||||
|
}
|
||||||
|
entry, ok := catalog.Lookup(normalized)
|
||||||
|
if !ok {
|
||||||
|
return ResolvedSessionArtifact{}, fmt.Errorf("unsupported artifact source %q", source)
|
||||||
|
}
|
||||||
|
if !entry.Available {
|
||||||
|
return ResolvedSessionArtifact{}, &SessionArtifactNotFoundError{ArtifactID: normalized}
|
||||||
|
}
|
||||||
|
if err := validateResolvedContent(entry.Path, contentText); err != nil {
|
||||||
|
return ResolvedSessionArtifact{}, fmt.Errorf("validate %q: %w", normalized, err)
|
||||||
|
}
|
||||||
|
return ResolvedSessionArtifact{
|
||||||
|
ID: normalized,
|
||||||
|
Path: filepath.Clean(entry.Path),
|
||||||
|
ProducerStage: entry.ProducerStage,
|
||||||
|
OutputKind: entry.OutputKind,
|
||||||
|
Provenance: entry.Provenance,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
func manifestArtifactCandidates(paths SessionPaths, m *manifest.Manifest, spec artifactSpec) []ResolvedSessionArtifact {
|
func manifestArtifactCandidates(paths SessionPaths, m *manifest.Manifest, spec artifactSpec) []ResolvedSessionArtifact {
|
||||||
if m == nil || len(m.Stages) == 0 || spec.ProducerStage == "" || spec.OutputKind == "" {
|
if m == nil || len(m.Stages) == 0 || spec.ProducerStage == "" || spec.OutputKind == "" {
|
||||||
return nil
|
return nil
|
||||||
|
|||||||
@@ -18,9 +18,10 @@ func TestNormalizeSessionArtifactSource(t *testing.T) {
|
|||||||
wantID string
|
wantID string
|
||||||
wantErr string
|
wantErr string
|
||||||
}{
|
}{
|
||||||
{name: "legacy alias processed", source: "processed_transcript", wantID: ArtifactTranscriptPolished},
|
{name: "legacy alias processed unsupported", source: "processed_transcript", wantErr: "unsupported artifact source"},
|
||||||
{name: "legacy alias normalized", source: "normalized_transcript", wantID: ArtifactTranscriptFull},
|
{name: "legacy alias normalized unsupported", source: "normalized_transcript", wantErr: "unsupported artifact source"},
|
||||||
{name: "legacy alias trimmed", source: "trimmed_transcript", wantID: ArtifactTranscriptTrimmed},
|
{name: "legacy alias trimmed unsupported", source: "trimmed_transcript", wantErr: "unsupported artifact source"},
|
||||||
|
{name: "configured source unsupported in built-in normalization", source: "narratio.artifact.session_recap", wantErr: "unsupported artifact source"},
|
||||||
{name: "canonical", source: ArtifactTranscriptTrimmed, wantID: ArtifactTranscriptTrimmed},
|
{name: "canonical", source: ArtifactTranscriptTrimmed, wantID: ArtifactTranscriptTrimmed},
|
||||||
{name: "unsupported", source: "narratio.unknown", wantErr: "unsupported artifact source"},
|
{name: "unsupported", source: "narratio.unknown", wantErr: "unsupported artifact source"},
|
||||||
}
|
}
|
||||||
@@ -67,7 +68,7 @@ func TestResolveSessionArtifactPrefersManifestOutput(t *testing.T) {
|
|||||||
{Kind: "transcript_normalized", LocalPath: manifestPath, ProducerRunID: "run-123"},
|
{Kind: "transcript_normalized", LocalPath: manifestPath, ProducerRunID: "run-123"},
|
||||||
})
|
})
|
||||||
|
|
||||||
resolved, err := ResolveSessionArtifact(paths, m, "normalized_transcript")
|
resolved, err := ResolveSessionArtifact(paths, m, ArtifactTranscriptFull)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("ResolveSessionArtifact() error = %v", err)
|
t.Fatalf("ResolveSessionArtifact() error = %v", err)
|
||||||
}
|
}
|
||||||
@@ -137,3 +138,131 @@ func TestResolveSessionArtifactValidatesTranscriptShape(t *testing.T) {
|
|||||||
t.Fatalf("error = %q, want segments validation error", err.Error())
|
t.Fatalf("error = %q, want segments validation error", err.Error())
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestResolveSessionArtifactWithCatalogBuiltInBehaviorUnchanged(t *testing.T) {
|
||||||
|
workspace := t.TempDir()
|
||||||
|
paths := buildSessionPaths(workspace, "campaign", "session")
|
||||||
|
canonicalPath := filepath.Join(paths.TranscriptsDir, "trimmed.json")
|
||||||
|
if err := os.MkdirAll(filepath.Dir(canonicalPath), 0o755); err != nil {
|
||||||
|
t.Fatalf("MkdirAll() error = %v", err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(canonicalPath, []byte(`{"segments":[]}`), 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
resolved, err := ResolveSessionArtifactWithCatalog(paths, nil, ArtifactTranscriptTrimmed, NewArtifactCatalog())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ResolveSessionArtifactWithCatalog() error = %v", err)
|
||||||
|
}
|
||||||
|
if resolved.Path != canonicalPath {
|
||||||
|
t.Fatalf("resolved.Path = %q, want %q", resolved.Path, canonicalPath)
|
||||||
|
}
|
||||||
|
if resolved.Provenance != "fallback.canonical_path" {
|
||||||
|
t.Fatalf("provenance = %q, want fallback.canonical_path", resolved.Provenance)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestResolveSessionArtifactWithCatalogConfiguredAvailableGenerated(t *testing.T) {
|
||||||
|
workspace := t.TempDir()
|
||||||
|
paths := buildSessionPaths(workspace, "campaign", "session")
|
||||||
|
outputPath := filepath.Join(paths.ArtifactsDir, "session_recap.md")
|
||||||
|
if err := os.MkdirAll(filepath.Dir(outputPath), 0o755); err != nil {
|
||||||
|
t.Fatalf("MkdirAll() error = %v", err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(outputPath, []byte("recap\n"), 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
catalog := NewArtifactCatalog()
|
||||||
|
if err := catalog.RegisterConfiguredArtifacts(
|
||||||
|
map[string]ConfiguredArtifactDefinition{
|
||||||
|
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
|
||||||
|
},
|
||||||
|
nil,
|
||||||
|
); err != nil {
|
||||||
|
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
|
||||||
|
}
|
||||||
|
sourceID := ConfiguredArtifactSourceID("session_recap")
|
||||||
|
if err := catalog.MarkAvailableGenerated(sourceID, outputPath); err != nil {
|
||||||
|
t.Fatalf("MarkAvailableGenerated() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
resolved, err := ResolveSessionArtifactWithCatalog(paths, nil, sourceID, catalog)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ResolveSessionArtifactWithCatalog() error = %v", err)
|
||||||
|
}
|
||||||
|
if resolved.Path != outputPath {
|
||||||
|
t.Fatalf("resolved.Path = %q, want %q", resolved.Path, outputPath)
|
||||||
|
}
|
||||||
|
if resolved.Provenance != ArtifactProvenanceGeneratedCurrentAnalyzeRun {
|
||||||
|
t.Fatalf("provenance = %q, want %q", resolved.Provenance, ArtifactProvenanceGeneratedCurrentAnalyzeRun)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestResolveSessionArtifactWithCatalogConfiguredAvailableFromDisk(t *testing.T) {
|
||||||
|
workspace := t.TempDir()
|
||||||
|
paths := buildSessionPaths(workspace, "campaign", "session")
|
||||||
|
outputPath := filepath.Join(paths.ArtifactsDir, "player_handout.md")
|
||||||
|
if err := os.MkdirAll(filepath.Dir(outputPath), 0o755); err != nil {
|
||||||
|
t.Fatalf("MkdirAll() error = %v", err)
|
||||||
|
}
|
||||||
|
if err := os.WriteFile(outputPath, []byte("handout\n"), 0o644); err != nil {
|
||||||
|
t.Fatalf("WriteFile() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
catalog := NewArtifactCatalog()
|
||||||
|
if err := catalog.RegisterConfiguredArtifacts(
|
||||||
|
map[string]ConfiguredArtifactDefinition{
|
||||||
|
"player_handout": {Enabled: false, OutputPath: "artifacts/player_handout.md"},
|
||||||
|
},
|
||||||
|
nil,
|
||||||
|
); err != nil {
|
||||||
|
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
|
||||||
|
}
|
||||||
|
sourceID := ConfiguredArtifactSourceID("player_handout")
|
||||||
|
if err := catalog.MarkAvailableFromDisk(sourceID, outputPath); err != nil {
|
||||||
|
t.Fatalf("MarkAvailableFromDisk() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
resolved, err := ResolveSessionArtifactWithCatalog(paths, nil, sourceID, catalog)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("ResolveSessionArtifactWithCatalog() error = %v", err)
|
||||||
|
}
|
||||||
|
if resolved.Provenance != ArtifactProvenanceDisabledFromDisk {
|
||||||
|
t.Fatalf("provenance = %q, want %q", resolved.Provenance, ArtifactProvenanceDisabledFromDisk)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestResolveSessionArtifactWithCatalogConfiguredPlannedButUnavailable(t *testing.T) {
|
||||||
|
workspace := t.TempDir()
|
||||||
|
paths := buildSessionPaths(workspace, "campaign", "session")
|
||||||
|
catalog := NewArtifactCatalog()
|
||||||
|
if err := catalog.RegisterConfiguredArtifacts(
|
||||||
|
map[string]ConfiguredArtifactDefinition{
|
||||||
|
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
|
||||||
|
},
|
||||||
|
nil,
|
||||||
|
); err != nil {
|
||||||
|
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
|
||||||
|
}
|
||||||
|
_, err := ResolveSessionArtifactWithCatalog(paths, nil, ConfiguredArtifactSourceID("session_recap"), catalog)
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error, got nil")
|
||||||
|
}
|
||||||
|
if !errors.Is(err, ErrSessionArtifactNotFound) {
|
||||||
|
t.Fatalf("errors.Is(err, ErrSessionArtifactNotFound)=false; err=%v", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestResolveSessionArtifactWithCatalogUnsupportedConfiguredSourceFails(t *testing.T) {
|
||||||
|
workspace := t.TempDir()
|
||||||
|
paths := buildSessionPaths(workspace, "campaign", "session")
|
||||||
|
catalog := NewArtifactCatalog()
|
||||||
|
_, err := ResolveSessionArtifactWithCatalog(paths, nil, "narratio.artifact.unknown", catalog)
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error, got nil")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), "unsupported artifact source") {
|
||||||
|
t.Fatalf("error = %q, want unsupported artifact source", err.Error())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
224
internal/artifacts/catalog.go
Normal file
224
internal/artifacts/catalog.go
Normal file
@@ -0,0 +1,224 @@
|
|||||||
|
package artifacts
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"sort"
|
||||||
|
"strings"
|
||||||
|
)
|
||||||
|
|
||||||
|
const (
|
||||||
|
ArtifactProvenanceGeneratedCurrentAnalyzeRun = "generated.current_analyze_run"
|
||||||
|
ArtifactProvenanceDisabledFromDisk = "filesystem.disabled_artifact_output"
|
||||||
|
)
|
||||||
|
|
||||||
|
// ConfiguredArtifactDefinition describes one configured analyze artifact.
|
||||||
|
type ConfiguredArtifactDefinition struct {
|
||||||
|
Enabled bool
|
||||||
|
OutputPath string
|
||||||
|
}
|
||||||
|
|
||||||
|
// CatalogEntry is one runtime catalog entry resolved by source ID.
|
||||||
|
type CatalogEntry struct {
|
||||||
|
SourceID string
|
||||||
|
ConfiguredKey string
|
||||||
|
CanonicalRelPath string
|
||||||
|
ProducerStage string
|
||||||
|
OutputKind string
|
||||||
|
Planned bool
|
||||||
|
Executable bool
|
||||||
|
Available bool
|
||||||
|
Path string
|
||||||
|
Provenance string
|
||||||
|
}
|
||||||
|
|
||||||
|
// ArtifactCatalog tracks built-in and configured artifact definitions and runtime state.
|
||||||
|
type ArtifactCatalog struct {
|
||||||
|
entries map[string]CatalogEntry
|
||||||
|
configuredIndex map[string]string
|
||||||
|
}
|
||||||
|
|
||||||
|
// NewArtifactCatalog returns an empty runtime artifact catalog.
|
||||||
|
func NewArtifactCatalog() *ArtifactCatalog {
|
||||||
|
return &ArtifactCatalog{
|
||||||
|
entries: map[string]CatalogEntry{},
|
||||||
|
configuredIndex: map[string]string{},
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ConfiguredArtifactSourceID converts a configured artifact key into canonical source ID.
|
||||||
|
func ConfiguredArtifactSourceID(key string) string {
|
||||||
|
return "narratio.artifact." + strings.TrimSpace(key)
|
||||||
|
}
|
||||||
|
|
||||||
|
// RegisterBuiltIns registers built-in source definitions used by runtime artifact resolution.
|
||||||
|
func (c *ArtifactCatalog) RegisterBuiltIns() error {
|
||||||
|
for _, id := range runtimeBuiltInArtifactIDs() {
|
||||||
|
spec, ok := artifactRegistry[id]
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("register built-ins: source %q not found in artifact registry", id)
|
||||||
|
}
|
||||||
|
if err := c.addEntry(CatalogEntry{
|
||||||
|
SourceID: spec.ID,
|
||||||
|
CanonicalRelPath: spec.CanonicalRelPath,
|
||||||
|
ProducerStage: spec.ProducerStage,
|
||||||
|
OutputKind: spec.OutputKind,
|
||||||
|
Planned: true,
|
||||||
|
}); err != nil {
|
||||||
|
return fmt.Errorf("register built-ins: %w", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// RegisterConfiguredArtifacts registers configured artifacts and applies executable selection.
|
||||||
|
func (c *ArtifactCatalog) RegisterConfiguredArtifacts(
|
||||||
|
configured map[string]ConfiguredArtifactDefinition,
|
||||||
|
selected []string,
|
||||||
|
) error {
|
||||||
|
keys := make([]string, 0, len(configured))
|
||||||
|
for key := range configured {
|
||||||
|
keys = append(keys, key)
|
||||||
|
}
|
||||||
|
sort.Strings(keys)
|
||||||
|
|
||||||
|
selectedSet := map[string]struct{}{}
|
||||||
|
for _, key := range selected {
|
||||||
|
trimmed := strings.TrimSpace(key)
|
||||||
|
if trimmed == "" {
|
||||||
|
return fmt.Errorf("selected artifact keys must be non-empty")
|
||||||
|
}
|
||||||
|
selectedSet[trimmed] = struct{}{}
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, key := range keys {
|
||||||
|
trimmed := strings.TrimSpace(key)
|
||||||
|
if trimmed == "" {
|
||||||
|
return fmt.Errorf("configured artifact keys must be non-empty")
|
||||||
|
}
|
||||||
|
def := configured[key]
|
||||||
|
sourceID := ConfiguredArtifactSourceID(trimmed)
|
||||||
|
if _, exists := c.configuredIndex[trimmed]; exists {
|
||||||
|
return fmt.Errorf("duplicate configured artifact key %q", trimmed)
|
||||||
|
}
|
||||||
|
|
||||||
|
executable := def.Enabled
|
||||||
|
if len(selectedSet) > 0 {
|
||||||
|
_, executable = selectedSet[trimmed]
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := c.addEntry(CatalogEntry{
|
||||||
|
SourceID: sourceID,
|
||||||
|
ConfiguredKey: trimmed,
|
||||||
|
CanonicalRelPath: strings.TrimSpace(def.OutputPath),
|
||||||
|
ProducerStage: "analyze",
|
||||||
|
OutputKind: "scriptorium_artifact",
|
||||||
|
Planned: true,
|
||||||
|
Executable: executable,
|
||||||
|
}); err != nil {
|
||||||
|
return fmt.Errorf("register configured artifact %q: %w", trimmed, err)
|
||||||
|
}
|
||||||
|
c.configuredIndex[trimmed] = sourceID
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(selectedSet) > 0 {
|
||||||
|
for key := range selectedSet {
|
||||||
|
if _, ok := c.configuredIndex[key]; !ok {
|
||||||
|
return fmt.Errorf("selected artifact %q is not configured", key)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Lookup returns one catalog entry by source ID.
|
||||||
|
func (c *ArtifactCatalog) Lookup(sourceID string) (CatalogEntry, bool) {
|
||||||
|
if c == nil {
|
||||||
|
return CatalogEntry{}, false
|
||||||
|
}
|
||||||
|
entry, ok := c.entries[strings.TrimSpace(sourceID)]
|
||||||
|
return entry, ok
|
||||||
|
}
|
||||||
|
|
||||||
|
// SourceIDForConfiguredKey returns canonical source ID for one configured key.
|
||||||
|
func (c *ArtifactCatalog) SourceIDForConfiguredKey(key string) (string, bool) {
|
||||||
|
if c == nil {
|
||||||
|
return "", false
|
||||||
|
}
|
||||||
|
sourceID, ok := c.configuredIndex[strings.TrimSpace(key)]
|
||||||
|
return sourceID, ok
|
||||||
|
}
|
||||||
|
|
||||||
|
// ListConfigured returns configured entries sorted by configured key.
|
||||||
|
func (c *ArtifactCatalog) ListConfigured() []CatalogEntry {
|
||||||
|
if c == nil || len(c.configuredIndex) == 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
keys := make([]string, 0, len(c.configuredIndex))
|
||||||
|
for key := range c.configuredIndex {
|
||||||
|
keys = append(keys, key)
|
||||||
|
}
|
||||||
|
sort.Strings(keys)
|
||||||
|
out := make([]CatalogEntry, 0, len(keys))
|
||||||
|
for _, key := range keys {
|
||||||
|
sourceID := c.configuredIndex[key]
|
||||||
|
out = append(out, c.entries[sourceID])
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|
||||||
|
// MarkAvailableGenerated marks one source as available in current analyze execution.
|
||||||
|
func (c *ArtifactCatalog) MarkAvailableGenerated(sourceID, path string) error {
|
||||||
|
return c.markAvailable(sourceID, path, ArtifactProvenanceGeneratedCurrentAnalyzeRun)
|
||||||
|
}
|
||||||
|
|
||||||
|
// MarkAvailableFromDisk marks one source as available from disabled artifact on disk.
|
||||||
|
func (c *ArtifactCatalog) MarkAvailableFromDisk(sourceID, path string) error {
|
||||||
|
return c.markAvailable(sourceID, path, ArtifactProvenanceDisabledFromDisk)
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *ArtifactCatalog) markAvailable(sourceID, path, provenance string) error {
|
||||||
|
if c == nil {
|
||||||
|
return fmt.Errorf("artifact catalog is nil")
|
||||||
|
}
|
||||||
|
normalizedID := strings.TrimSpace(sourceID)
|
||||||
|
entry, ok := c.entries[normalizedID]
|
||||||
|
if !ok {
|
||||||
|
return fmt.Errorf("unknown artifact source %q", sourceID)
|
||||||
|
}
|
||||||
|
trimmedPath := strings.TrimSpace(path)
|
||||||
|
if trimmedPath == "" {
|
||||||
|
return fmt.Errorf("artifact path is required")
|
||||||
|
}
|
||||||
|
entry.Available = true
|
||||||
|
entry.Path = trimmedPath
|
||||||
|
entry.Provenance = provenance
|
||||||
|
c.entries[normalizedID] = entry
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func (c *ArtifactCatalog) addEntry(entry CatalogEntry) error {
|
||||||
|
if c == nil {
|
||||||
|
return fmt.Errorf("artifact catalog is nil")
|
||||||
|
}
|
||||||
|
sourceID := strings.TrimSpace(entry.SourceID)
|
||||||
|
if sourceID == "" {
|
||||||
|
return fmt.Errorf("source id is required")
|
||||||
|
}
|
||||||
|
if _, exists := c.entries[sourceID]; exists {
|
||||||
|
return fmt.Errorf("source id %q is already registered", sourceID)
|
||||||
|
}
|
||||||
|
entry.SourceID = sourceID
|
||||||
|
c.entries[sourceID] = entry
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func runtimeBuiltInArtifactIDs() []string {
|
||||||
|
return []string{
|
||||||
|
ArtifactTranscriptMerged,
|
||||||
|
ArtifactTranscriptPolished,
|
||||||
|
ArtifactTranscriptFull,
|
||||||
|
ArtifactTranscriptTrimmed,
|
||||||
|
ArtifactBoundsSession,
|
||||||
|
}
|
||||||
|
}
|
||||||
188
internal/artifacts/catalog_test.go
Normal file
188
internal/artifacts/catalog_test.go
Normal file
@@ -0,0 +1,188 @@
|
|||||||
|
package artifacts
|
||||||
|
|
||||||
|
import "testing"
|
||||||
|
|
||||||
|
func TestArtifactCatalogRegisterBuiltInsAndLookup(t *testing.T) {
|
||||||
|
catalog := NewArtifactCatalog()
|
||||||
|
if err := catalog.RegisterBuiltIns(); err != nil {
|
||||||
|
t.Fatalf("RegisterBuiltIns() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
entry, ok := catalog.Lookup(ArtifactTranscriptFull)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("Lookup(%q) ok = false, want true", ArtifactTranscriptFull)
|
||||||
|
}
|
||||||
|
if !entry.Planned {
|
||||||
|
t.Fatalf("entry.Planned = false, want true")
|
||||||
|
}
|
||||||
|
if entry.Executable {
|
||||||
|
t.Fatalf("entry.Executable = true, want false")
|
||||||
|
}
|
||||||
|
if entry.CanonicalRelPath != "transcripts/normalized.json" {
|
||||||
|
t.Fatalf("entry.CanonicalRelPath = %q, want transcripts/normalized.json", entry.CanonicalRelPath)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestArtifactCatalogRegisterConfiguredArtifactsDefaultsToEnabled(t *testing.T) {
|
||||||
|
catalog := NewArtifactCatalog()
|
||||||
|
if err := catalog.RegisterConfiguredArtifacts(map[string]ConfiguredArtifactDefinition{
|
||||||
|
"player_handout": {Enabled: false, OutputPath: "artifacts/player_handout.md"},
|
||||||
|
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
|
||||||
|
}, nil); err != nil {
|
||||||
|
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
recapID, ok := catalog.SourceIDForConfiguredKey("session_recap")
|
||||||
|
if !ok {
|
||||||
|
t.Fatal("SourceIDForConfiguredKey(session_recap) ok = false, want true")
|
||||||
|
}
|
||||||
|
recap, ok := catalog.Lookup(recapID)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("Lookup(%q) ok = false, want true", recapID)
|
||||||
|
}
|
||||||
|
if !recap.Executable {
|
||||||
|
t.Fatalf("recap.Executable = false, want true")
|
||||||
|
}
|
||||||
|
|
||||||
|
handoutID, ok := catalog.SourceIDForConfiguredKey("player_handout")
|
||||||
|
if !ok {
|
||||||
|
t.Fatal("SourceIDForConfiguredKey(player_handout) ok = false, want true")
|
||||||
|
}
|
||||||
|
handout, ok := catalog.Lookup(handoutID)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("Lookup(%q) ok = false, want true", handoutID)
|
||||||
|
}
|
||||||
|
if handout.Executable {
|
||||||
|
t.Fatalf("handout.Executable = true, want false")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestArtifactCatalogRegisterConfiguredArtifactsSelectedSetOverridesEnabled(t *testing.T) {
|
||||||
|
catalog := NewArtifactCatalog()
|
||||||
|
if err := catalog.RegisterConfiguredArtifacts(
|
||||||
|
map[string]ConfiguredArtifactDefinition{
|
||||||
|
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
|
||||||
|
"player_handout": {Enabled: false, OutputPath: "artifacts/player_handout.md"},
|
||||||
|
},
|
||||||
|
[]string{"player_handout"},
|
||||||
|
); err != nil {
|
||||||
|
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
entries := catalog.ListConfigured()
|
||||||
|
if len(entries) != 2 {
|
||||||
|
t.Fatalf("ListConfigured() len = %d, want 2", len(entries))
|
||||||
|
}
|
||||||
|
if entries[0].ConfiguredKey != "player_handout" || entries[0].Executable != true {
|
||||||
|
t.Fatalf("entries[0] = %+v, want player_handout executable", entries[0])
|
||||||
|
}
|
||||||
|
if entries[1].ConfiguredKey != "session_recap" || entries[1].Executable != false {
|
||||||
|
t.Fatalf("entries[1] = %+v, want session_recap disabled by selection", entries[1])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestArtifactCatalogRejectsSelectedUnknownArtifact(t *testing.T) {
|
||||||
|
catalog := NewArtifactCatalog()
|
||||||
|
err := catalog.RegisterConfiguredArtifacts(
|
||||||
|
map[string]ConfiguredArtifactDefinition{
|
||||||
|
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
|
||||||
|
},
|
||||||
|
[]string{"unknown"},
|
||||||
|
)
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("RegisterConfiguredArtifacts() error = nil, want non-nil")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestArtifactCatalogRejectsConfiguredSourceConflictAcrossRegistrations(t *testing.T) {
|
||||||
|
catalog := NewArtifactCatalog()
|
||||||
|
if err := catalog.RegisterConfiguredArtifacts(
|
||||||
|
map[string]ConfiguredArtifactDefinition{
|
||||||
|
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
|
||||||
|
},
|
||||||
|
nil,
|
||||||
|
); err != nil {
|
||||||
|
t.Fatalf("RegisterConfiguredArtifacts() first call error = %v", err)
|
||||||
|
}
|
||||||
|
err := catalog.RegisterConfiguredArtifacts(
|
||||||
|
map[string]ConfiguredArtifactDefinition{
|
||||||
|
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
|
||||||
|
},
|
||||||
|
nil,
|
||||||
|
)
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("RegisterConfiguredArtifacts() error = nil, want non-nil")
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestArtifactCatalogMarkAvailableGenerated(t *testing.T) {
|
||||||
|
catalog := NewArtifactCatalog()
|
||||||
|
if err := catalog.RegisterConfiguredArtifacts(
|
||||||
|
map[string]ConfiguredArtifactDefinition{
|
||||||
|
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
|
||||||
|
},
|
||||||
|
nil,
|
||||||
|
); err != nil {
|
||||||
|
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
sourceID, _ := catalog.SourceIDForConfiguredKey("session_recap")
|
||||||
|
if err := catalog.MarkAvailableGenerated(sourceID, "/tmp/session_recap.md"); err != nil {
|
||||||
|
t.Fatalf("MarkAvailableGenerated() error = %v", err)
|
||||||
|
}
|
||||||
|
entry, _ := catalog.Lookup(sourceID)
|
||||||
|
if !entry.Available {
|
||||||
|
t.Fatalf("entry.Available = false, want true")
|
||||||
|
}
|
||||||
|
if entry.Provenance != ArtifactProvenanceGeneratedCurrentAnalyzeRun {
|
||||||
|
t.Fatalf("entry.Provenance = %q, want %q", entry.Provenance, ArtifactProvenanceGeneratedCurrentAnalyzeRun)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestArtifactCatalogMarkAvailableFromDisk(t *testing.T) {
|
||||||
|
catalog := NewArtifactCatalog()
|
||||||
|
if err := catalog.RegisterConfiguredArtifacts(
|
||||||
|
map[string]ConfiguredArtifactDefinition{
|
||||||
|
"session_recap": {Enabled: false, OutputPath: "artifacts/session_recap.md"},
|
||||||
|
},
|
||||||
|
nil,
|
||||||
|
); err != nil {
|
||||||
|
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
sourceID, _ := catalog.SourceIDForConfiguredKey("session_recap")
|
||||||
|
if err := catalog.MarkAvailableFromDisk(sourceID, "/tmp/session_recap.md"); err != nil {
|
||||||
|
t.Fatalf("MarkAvailableFromDisk() error = %v", err)
|
||||||
|
}
|
||||||
|
entry, _ := catalog.Lookup(sourceID)
|
||||||
|
if !entry.Available {
|
||||||
|
t.Fatalf("entry.Available = false, want true")
|
||||||
|
}
|
||||||
|
if entry.Provenance != ArtifactProvenanceDisabledFromDisk {
|
||||||
|
t.Fatalf("entry.Provenance = %q, want %q", entry.Provenance, ArtifactProvenanceDisabledFromDisk)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestArtifactCatalogLookupPlannedButUnavailable(t *testing.T) {
|
||||||
|
catalog := NewArtifactCatalog()
|
||||||
|
if err := catalog.RegisterConfiguredArtifacts(
|
||||||
|
map[string]ConfiguredArtifactDefinition{
|
||||||
|
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
|
||||||
|
},
|
||||||
|
nil,
|
||||||
|
); err != nil {
|
||||||
|
t.Fatalf("RegisterConfiguredArtifacts() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
sourceID, _ := catalog.SourceIDForConfiguredKey("session_recap")
|
||||||
|
entry, ok := catalog.Lookup(sourceID)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("Lookup(%q) ok = false, want true", sourceID)
|
||||||
|
}
|
||||||
|
if !entry.Planned {
|
||||||
|
t.Fatalf("entry.Planned = false, want true")
|
||||||
|
}
|
||||||
|
if entry.Available {
|
||||||
|
t.Fatalf("entry.Available = true, want false")
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -176,6 +176,7 @@ type ScriptoriumConfig struct {
|
|||||||
// ScriptoriumArtifactConfig configures one named output artifact workflow.
|
// ScriptoriumArtifactConfig configures one named output artifact workflow.
|
||||||
type ScriptoriumArtifactConfig struct {
|
type ScriptoriumArtifactConfig struct {
|
||||||
Enabled bool `yaml:"enabled"`
|
Enabled bool `yaml:"enabled"`
|
||||||
|
DependsOn []string `yaml:"depends_on"`
|
||||||
RenderDebug *bool `yaml:"render_debug"`
|
RenderDebug *bool `yaml:"render_debug"`
|
||||||
PromptID string `yaml:"prompt_id"`
|
PromptID string `yaml:"prompt_id"`
|
||||||
ProfileID string `yaml:"profile_id"`
|
ProfileID string `yaml:"profile_id"`
|
||||||
|
|||||||
@@ -11,6 +11,7 @@ const (
|
|||||||
DefaultS3AccessKeyIDEnv = "OBJECT_STORAGE_KEY_ID"
|
DefaultS3AccessKeyIDEnv = "OBJECT_STORAGE_KEY_ID"
|
||||||
DefaultS3SecretAccessKeyEnv = "OBJECT_STORAGE_KEY"
|
DefaultS3SecretAccessKeyEnv = "OBJECT_STORAGE_KEY"
|
||||||
DefaultStorageS3RootPrefix = "dnd"
|
DefaultStorageS3RootPrefix = "dnd"
|
||||||
|
DefaultWorkspaceRoot = "/var/lib/narratio"
|
||||||
DefaultSpoolRoot = "/var/spool/narratio"
|
DefaultSpoolRoot = "/var/spool/narratio"
|
||||||
|
|
||||||
DefaultWhisperXLanguage = "en"
|
DefaultWhisperXLanguage = "en"
|
||||||
@@ -31,6 +32,7 @@ const (
|
|||||||
|
|
||||||
DefaultScriptoriumBinary = "scriptorium"
|
DefaultScriptoriumBinary = "scriptorium"
|
||||||
DefaultScriptoriumTimeout = "10m"
|
DefaultScriptoriumTimeout = "10m"
|
||||||
|
DefaultScriptoriumArtifactOutputRoot = "artifacts"
|
||||||
|
|
||||||
DefaultTrimBoundsTimeout = "10m"
|
DefaultTrimBoundsTimeout = "10m"
|
||||||
DefaultTrimSeriatimReport = false
|
DefaultTrimSeriatimReport = false
|
||||||
|
|||||||
@@ -152,6 +152,7 @@ func applyPipelineDefaults(cfg *PipelineConfig) {
|
|||||||
if cfg == nil {
|
if cfg == nil {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
applyWorkspaceDefaults(&cfg.Workspace)
|
||||||
applyStorageDefaults(&cfg.Storage)
|
applyStorageDefaults(&cfg.Storage)
|
||||||
applySpoolDefaults(&cfg.Spool)
|
applySpoolDefaults(&cfg.Spool)
|
||||||
applyArchiveDefaults(&cfg.Archive)
|
applyArchiveDefaults(&cfg.Archive)
|
||||||
@@ -166,6 +167,15 @@ func applyPipelineDefaults(cfg *PipelineConfig) {
|
|||||||
applyScriptoriumDefaults(cfg.Scriptorium)
|
applyScriptoriumDefaults(cfg.Scriptorium)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func applyWorkspaceDefaults(cfg *WorkspaceConfig) {
|
||||||
|
if cfg == nil {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
if cfg.Root == "" {
|
||||||
|
cfg.Root = DefaultWorkspaceRoot
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func applyStorageDefaults(cfg *StorageConfig) {
|
func applyStorageDefaults(cfg *StorageConfig) {
|
||||||
if cfg == nil {
|
if cfg == nil {
|
||||||
return
|
return
|
||||||
|
|||||||
@@ -15,6 +15,7 @@ func TestLoadAndValidate(t *testing.T) {
|
|||||||
wantLoadErr string
|
wantLoadErr string
|
||||||
wantValidate string
|
wantValidate string
|
||||||
checkDefault bool
|
checkDefault bool
|
||||||
|
wantRoot string
|
||||||
}{
|
}{
|
||||||
{
|
{
|
||||||
name: "valid minimal config",
|
name: "valid minimal config",
|
||||||
@@ -39,6 +40,7 @@ inputs:
|
|||||||
glossary_file: ./glossary.yml
|
glossary_file: ./glossary.yml
|
||||||
`,
|
`,
|
||||||
checkDefault: true,
|
checkDefault: true,
|
||||||
|
wantRoot: "/tmp/narratio",
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
name: "seriatim and audita sections can be omitted",
|
name: "seriatim and audita sections can be omitted",
|
||||||
@@ -59,6 +61,26 @@ inputs:
|
|||||||
glossary_file: ./glossary.yml
|
glossary_file: ./glossary.yml
|
||||||
`,
|
`,
|
||||||
checkDefault: true,
|
checkDefault: true,
|
||||||
|
wantRoot: "/tmp/narratio",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "workspace root defaults when omitted",
|
||||||
|
pipelineYAML: `whisperx:
|
||||||
|
transcribe_url: https://transcription.ai.rakestrawhome.com/transcribe
|
||||||
|
analyzer:
|
||||||
|
timeout: 20m
|
||||||
|
notification:
|
||||||
|
timeout: 15s
|
||||||
|
`,
|
||||||
|
sessionYAML: `session_id: 2026-05-03
|
||||||
|
inputs:
|
||||||
|
audio_dir: ./audio
|
||||||
|
speakers_file: ./speakers.yml
|
||||||
|
autocorrect_file: ./autocorrect.yml
|
||||||
|
glossary_file: ./glossary.yml
|
||||||
|
`,
|
||||||
|
checkDefault: true,
|
||||||
|
wantRoot: DefaultWorkspaceRoot,
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
name: "unknown pipeline field fails",
|
name: "unknown pipeline field fails",
|
||||||
@@ -713,6 +735,9 @@ inputs:
|
|||||||
t.Fatalf("SessionPath = %q, want %q", cfg.SessionPath, sessionPath)
|
t.Fatalf("SessionPath = %q, want %q", cfg.SessionPath, sessionPath)
|
||||||
}
|
}
|
||||||
if tt.checkDefault {
|
if tt.checkDefault {
|
||||||
|
if tt.wantRoot != "" && cfg.Pipeline.Workspace.Root != tt.wantRoot {
|
||||||
|
t.Fatalf("workspace.root = %q, want %q", cfg.Pipeline.Workspace.Root, tt.wantRoot)
|
||||||
|
}
|
||||||
if cfg.Pipeline.WhisperX.Language != "en" {
|
if cfg.Pipeline.WhisperX.Language != "en" {
|
||||||
t.Fatalf("whisperx.language = %q, want %q", cfg.Pipeline.WhisperX.Language, "en")
|
t.Fatalf("whisperx.language = %q, want %q", cfg.Pipeline.WhisperX.Language, "en")
|
||||||
}
|
}
|
||||||
@@ -866,15 +891,59 @@ func TestValidateMissingAudioSource(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
func TestExamplesLoadAndValidate(t *testing.T) {
|
func TestExamplesLoadAndValidate(t *testing.T) {
|
||||||
pipelinePath := filepath.Join("..", "..", "examples", "pipeline.minimal.yml")
|
examplesDir := filepath.Join("..", "..", "examples")
|
||||||
sessionPath := filepath.Join("..", "..", "examples", "session.minimal.yml")
|
tests := []struct {
|
||||||
|
name string
|
||||||
|
pipelineFile string
|
||||||
|
sessionFile string
|
||||||
|
sessionOpts SessionLoadOptions
|
||||||
|
}{
|
||||||
|
{
|
||||||
|
name: "minimal pipeline with local audio session",
|
||||||
|
pipelineFile: "pipeline.minimal.yml",
|
||||||
|
sessionFile: "session.local-audio.yml",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "production pipeline with s3 audio session",
|
||||||
|
pipelineFile: "pipeline.production.yml",
|
||||||
|
sessionFile: "session.s3-audio.yml",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "full annotated pipeline with local audio session",
|
||||||
|
pipelineFile: "pipeline.full.annotated.yml",
|
||||||
|
sessionFile: "session.local-audio.yml",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "template session renders with session_id option",
|
||||||
|
pipelineFile: "pipeline.minimal.yml",
|
||||||
|
sessionFile: "session.template.yml",
|
||||||
|
sessionOpts: SessionLoadOptions{
|
||||||
|
SessionID: "2026-05-03",
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
cfg, err := Load(pipelinePath, sessionPath)
|
for _, tt := range tests {
|
||||||
|
t.Run(tt.name, func(t *testing.T) {
|
||||||
|
pipelinePath := filepath.Join(examplesDir, tt.pipelineFile)
|
||||||
|
sessionPath := filepath.Join(examplesDir, tt.sessionFile)
|
||||||
|
|
||||||
|
var (
|
||||||
|
cfg *Config
|
||||||
|
err error
|
||||||
|
)
|
||||||
|
if strings.TrimSpace(tt.sessionOpts.SessionID) == "" {
|
||||||
|
cfg, err = Load(pipelinePath, sessionPath)
|
||||||
|
} else {
|
||||||
|
cfg, err = LoadWithSessionOptions(pipelinePath, sessionPath, tt.sessionOpts)
|
||||||
|
}
|
||||||
if err != nil {
|
if err != nil {
|
||||||
t.Fatalf("Load(examples) error = %v", err)
|
t.Fatalf("load example config error = %v", err)
|
||||||
}
|
}
|
||||||
if err := Validate(cfg); err != nil {
|
if err := Validate(cfg); err != nil {
|
||||||
t.Fatalf("Validate(examples) error = %v", err)
|
t.Fatalf("validate example config error = %v", err)
|
||||||
|
}
|
||||||
|
})
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -24,7 +24,7 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
|
|||||||
output_path: artifacts/session_recap.md
|
output_path: artifacts/session_recap.md
|
||||||
inputs:
|
inputs:
|
||||||
transcript:
|
transcript:
|
||||||
source: processed_transcript
|
source: narratio.transcript.polished
|
||||||
required: true
|
required: true
|
||||||
vars:
|
vars:
|
||||||
session_id: true
|
session_id: true
|
||||||
@@ -104,7 +104,7 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
|
|||||||
output_path: artifacts/session_recap.md
|
output_path: artifacts/session_recap.md
|
||||||
inputs:
|
inputs:
|
||||||
transcript:
|
transcript:
|
||||||
source: processed_transcript
|
source: narratio.transcript.polished
|
||||||
required: true
|
required: true
|
||||||
previous_recap:
|
previous_recap:
|
||||||
source: previous_session_artifact
|
source: previous_session_artifact
|
||||||
@@ -160,7 +160,7 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
|
|||||||
output_path: artifacts/session_recap.md
|
output_path: artifacts/session_recap.md
|
||||||
inputs:
|
inputs:
|
||||||
transcript:
|
transcript:
|
||||||
source: processed_transcript
|
source: narratio.transcript.polished
|
||||||
required: true
|
required: true
|
||||||
`,
|
`,
|
||||||
assert: func(t *testing.T, cfg *Config) {
|
assert: func(t *testing.T, cfg *Config) {
|
||||||
@@ -182,7 +182,7 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
|
|||||||
output_path: artifacts/session_recap.md
|
output_path: artifacts/session_recap.md
|
||||||
inputs:
|
inputs:
|
||||||
transcript:
|
transcript:
|
||||||
source: processed_transcript
|
source: narratio.transcript.polished
|
||||||
required: true
|
required: true
|
||||||
player_summary:
|
player_summary:
|
||||||
enabled: true
|
enabled: true
|
||||||
@@ -192,7 +192,7 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
|
|||||||
timeout: 3m
|
timeout: 3m
|
||||||
inputs:
|
inputs:
|
||||||
transcript:
|
transcript:
|
||||||
source: processed_transcript
|
source: narratio.transcript.polished
|
||||||
required: true
|
required: true
|
||||||
`,
|
`,
|
||||||
assert: func(t *testing.T, cfg *Config) {
|
assert: func(t *testing.T, cfg *Config) {
|
||||||
@@ -205,6 +205,195 @@ func TestScriptoriumLoadAndValidate(t *testing.T) {
|
|||||||
}
|
}
|
||||||
},
|
},
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
name: "valid artifact dependency is accepted",
|
||||||
|
scriptoriumYAML: `scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
artifacts:
|
||||||
|
session_recap:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd.session_recap
|
||||||
|
output_path: artifacts/session_recap.md
|
||||||
|
inputs:
|
||||||
|
transcript:
|
||||||
|
source: narratio.transcript.trimmed
|
||||||
|
required: true
|
||||||
|
player_handout:
|
||||||
|
enabled: true
|
||||||
|
depends_on:
|
||||||
|
- session_recap
|
||||||
|
prompt_id: dnd.player_handout
|
||||||
|
output_path: artifacts/player_handout.md
|
||||||
|
inputs:
|
||||||
|
recap:
|
||||||
|
source: narratio.artifact.session_recap
|
||||||
|
required: true
|
||||||
|
`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "valid dependency on disabled artifact with output path is accepted",
|
||||||
|
scriptoriumYAML: `scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
artifacts:
|
||||||
|
session_recap:
|
||||||
|
enabled: false
|
||||||
|
output_path: artifacts/session_recap.md
|
||||||
|
player_handout:
|
||||||
|
enabled: true
|
||||||
|
depends_on:
|
||||||
|
- session_recap
|
||||||
|
prompt_id: dnd.player_handout
|
||||||
|
output_path: artifacts/player_handout.md
|
||||||
|
inputs:
|
||||||
|
recap:
|
||||||
|
source: narratio.artifact.session_recap
|
||||||
|
required: true
|
||||||
|
`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "invalid artifact name fails validation",
|
||||||
|
scriptoriumYAML: `scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
artifacts:
|
||||||
|
SessionRecap:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd.session_recap
|
||||||
|
output_path: artifacts/session_recap.md
|
||||||
|
`,
|
||||||
|
wantValidateErr: "pipeline.scriptorium.artifacts keys must match ^[a-z][a-z0-9_]*$",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "artifact output path outside artifacts root fails validation",
|
||||||
|
scriptoriumYAML: `scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
artifacts:
|
||||||
|
session_recap:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd.session_recap
|
||||||
|
output_path: transcripts/session_recap.md
|
||||||
|
`,
|
||||||
|
wantValidateErr: "pipeline.scriptorium.artifacts.session_recap.output_path must be under artifacts/",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "missing depends_on for artifact source fails validation",
|
||||||
|
scriptoriumYAML: `scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
artifacts:
|
||||||
|
session_recap:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd.session_recap
|
||||||
|
output_path: artifacts/session_recap.md
|
||||||
|
inputs:
|
||||||
|
transcript:
|
||||||
|
source: narratio.transcript.polished
|
||||||
|
required: true
|
||||||
|
player_handout:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd.player_handout
|
||||||
|
output_path: artifacts/player_handout.md
|
||||||
|
inputs:
|
||||||
|
recap:
|
||||||
|
source: narratio.artifact.session_recap
|
||||||
|
required: true
|
||||||
|
`,
|
||||||
|
wantValidateErr: `pipeline.scriptorium.artifacts.player_handout.inputs.recap.source "narratio.artifact.session_recap" requires depends_on entry "session_recap"`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "dependency on unknown artifact fails validation",
|
||||||
|
scriptoriumYAML: `scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
artifacts:
|
||||||
|
player_handout:
|
||||||
|
enabled: true
|
||||||
|
depends_on:
|
||||||
|
- session_recap
|
||||||
|
prompt_id: dnd.player_handout
|
||||||
|
output_path: artifacts/player_handout.md
|
||||||
|
inputs:
|
||||||
|
recap:
|
||||||
|
source: narratio.artifact.session_recap
|
||||||
|
required: true
|
||||||
|
`,
|
||||||
|
wantValidateErr: `pipeline.scriptorium.artifacts.player_handout.depends_on[0] "session_recap" is not a configured artifact key`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "self dependency fails validation",
|
||||||
|
scriptoriumYAML: `scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
artifacts:
|
||||||
|
session_recap:
|
||||||
|
enabled: true
|
||||||
|
depends_on:
|
||||||
|
- session_recap
|
||||||
|
prompt_id: dnd.session_recap
|
||||||
|
output_path: artifacts/session_recap.md
|
||||||
|
`,
|
||||||
|
wantValidateErr: "pipeline.scriptorium.artifacts.session_recap.depends_on must not include itself",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "enabled dependency cycle fails validation",
|
||||||
|
scriptoriumYAML: `scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
artifacts:
|
||||||
|
artifact_a:
|
||||||
|
enabled: true
|
||||||
|
depends_on:
|
||||||
|
- artifact_b
|
||||||
|
prompt_id: dnd.a
|
||||||
|
output_path: artifacts/a.md
|
||||||
|
inputs:
|
||||||
|
b:
|
||||||
|
source: narratio.artifact.artifact_b
|
||||||
|
required: true
|
||||||
|
artifact_b:
|
||||||
|
enabled: true
|
||||||
|
depends_on:
|
||||||
|
- artifact_a
|
||||||
|
prompt_id: dnd.b
|
||||||
|
output_path: artifacts/b.md
|
||||||
|
inputs:
|
||||||
|
a:
|
||||||
|
source: narratio.artifact.artifact_a
|
||||||
|
required: true
|
||||||
|
`,
|
||||||
|
wantValidateErr: "pipeline.scriptorium.artifacts enabled dependencies must not contain cycles",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "artifact source typo fails validation",
|
||||||
|
scriptoriumYAML: `scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
artifacts:
|
||||||
|
player_handout:
|
||||||
|
enabled: true
|
||||||
|
prompt_id: dnd.player_handout
|
||||||
|
output_path: artifacts/player_handout.md
|
||||||
|
inputs:
|
||||||
|
recap:
|
||||||
|
source: narratio.artifact.session-recap
|
||||||
|
required: true
|
||||||
|
`,
|
||||||
|
wantValidateErr: `pipeline.scriptorium.artifacts.player_handout.inputs.recap.source "narratio.artifact.session-recap" is unsupported`,
|
||||||
|
},
|
||||||
|
{
|
||||||
|
name: "referenced disabled artifact missing output path fails validation",
|
||||||
|
scriptoriumYAML: `scriptorium:
|
||||||
|
binary: scriptorium
|
||||||
|
artifacts:
|
||||||
|
session_recap:
|
||||||
|
enabled: false
|
||||||
|
player_handout:
|
||||||
|
enabled: true
|
||||||
|
depends_on:
|
||||||
|
- session_recap
|
||||||
|
prompt_id: dnd.player_handout
|
||||||
|
output_path: artifacts/player_handout.md
|
||||||
|
inputs:
|
||||||
|
recap:
|
||||||
|
source: narratio.artifact.session_recap
|
||||||
|
required: true
|
||||||
|
`,
|
||||||
|
wantValidateErr: "pipeline.scriptorium.artifacts.session_recap.output_path is required when artifact is referenced",
|
||||||
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
for _, tt := range tests {
|
for _, tt := range tests {
|
||||||
@@ -258,6 +447,7 @@ audita:
|
|||||||
`
|
`
|
||||||
|
|
||||||
const testSessionBaseYAML = `session_id: 2026-05-03
|
const testSessionBaseYAML = `session_id: 2026-05-03
|
||||||
|
campaign: test-campaign
|
||||||
inputs:
|
inputs:
|
||||||
audio_dir: ./audio
|
audio_dir: ./audio
|
||||||
speakers_file: ./speakers.yml
|
speakers_file: ./speakers.yml
|
||||||
|
|||||||
@@ -324,20 +324,52 @@ func validateScriptorium(cfg *ScriptoriumConfig) error {
|
|||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
for artifactName, artifactCfg := range cfg.Artifacts {
|
configuredArtifacts := make(map[string]struct{}, len(cfg.Artifacts))
|
||||||
trimmedArtifactName := strings.TrimSpace(artifactName)
|
referencedArtifacts := make(map[string]struct{})
|
||||||
if trimmedArtifactName == "" {
|
for artifactName := range cfg.Artifacts {
|
||||||
return fmt.Errorf("pipeline.scriptorium.artifacts keys must be non-empty")
|
if !scriptoriumArtifactKeyRE.MatchString(strings.TrimSpace(artifactName)) {
|
||||||
|
return fmt.Errorf("pipeline.scriptorium.artifacts keys must match ^[a-z][a-z0-9_]*$")
|
||||||
}
|
}
|
||||||
|
configuredArtifacts[artifactName] = struct{}{}
|
||||||
|
}
|
||||||
|
|
||||||
|
for artifactName, artifactCfg := range cfg.Artifacts {
|
||||||
if artifactCfg.Enabled && strings.TrimSpace(artifactCfg.PromptID) == "" {
|
if artifactCfg.Enabled && strings.TrimSpace(artifactCfg.PromptID) == "" {
|
||||||
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.prompt_id is required when enabled", artifactName)
|
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.prompt_id is required when enabled", artifactName)
|
||||||
}
|
}
|
||||||
if artifactCfg.Enabled && strings.TrimSpace(artifactCfg.OutputPath) == "" {
|
if artifactCfg.Enabled && strings.TrimSpace(artifactCfg.OutputPath) == "" {
|
||||||
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.output_path is required when enabled", artifactName)
|
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.output_path is required when enabled", artifactName)
|
||||||
}
|
}
|
||||||
|
if strings.TrimSpace(artifactCfg.OutputPath) != "" {
|
||||||
|
pathField := "pipeline.scriptorium.artifacts." + artifactName + ".output_path"
|
||||||
|
if err := validateRelativeSafePath(pathField, artifactCfg.OutputPath); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if err := validatePathWithinRoot(pathField, artifactCfg.OutputPath, DefaultScriptoriumArtifactOutputRoot); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
if err := validateDuration("pipeline.scriptorium.artifacts."+artifactName+".timeout", artifactCfg.Timeout); err != nil {
|
if err := validateDuration("pipeline.scriptorium.artifacts."+artifactName+".timeout", artifactCfg.Timeout); err != nil {
|
||||||
return err
|
return err
|
||||||
}
|
}
|
||||||
|
|
||||||
|
depSet := make(map[string]struct{}, len(artifactCfg.DependsOn))
|
||||||
|
for i, depName := range artifactCfg.DependsOn {
|
||||||
|
trimmedDep := strings.TrimSpace(depName)
|
||||||
|
field := fmt.Sprintf("pipeline.scriptorium.artifacts.%s.depends_on[%d]", artifactName, i)
|
||||||
|
if trimmedDep == "" {
|
||||||
|
return fmt.Errorf("%s must be non-empty", field)
|
||||||
|
}
|
||||||
|
if _, ok := configuredArtifacts[trimmedDep]; !ok {
|
||||||
|
return fmt.Errorf("%s %q is not a configured artifact key", field, depName)
|
||||||
|
}
|
||||||
|
if trimmedDep == artifactName {
|
||||||
|
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.depends_on must not include itself", artifactName)
|
||||||
|
}
|
||||||
|
depSet[trimmedDep] = struct{}{}
|
||||||
|
referencedArtifacts[trimmedDep] = struct{}{}
|
||||||
|
}
|
||||||
|
|
||||||
for inputName, inputCfg := range artifactCfg.Inputs {
|
for inputName, inputCfg := range artifactCfg.Inputs {
|
||||||
trimmedInputName := strings.TrimSpace(inputName)
|
trimmedInputName := strings.TrimSpace(inputName)
|
||||||
if trimmedInputName == "" {
|
if trimmedInputName == "" {
|
||||||
@@ -347,8 +379,22 @@ func validateScriptorium(cfg *ScriptoriumConfig) error {
|
|||||||
if source == "" {
|
if source == "" {
|
||||||
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.inputs.%s.source is required", artifactName, inputName)
|
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.inputs.%s.source is required", artifactName, inputName)
|
||||||
}
|
}
|
||||||
if !isSupportedScriptoriumInputSource(source) {
|
|
||||||
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.inputs.%s.source %q is unsupported", artifactName, inputName, inputCfg.Source)
|
referencedArtifact, err := validateScriptoriumInputSource(artifactName, inputName, source, configuredArtifacts)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if referencedArtifact != "" {
|
||||||
|
if _, ok := depSet[referencedArtifact]; !ok {
|
||||||
|
return fmt.Errorf(
|
||||||
|
"pipeline.scriptorium.artifacts.%s.inputs.%s.source %q requires depends_on entry %q",
|
||||||
|
artifactName,
|
||||||
|
inputName,
|
||||||
|
source,
|
||||||
|
referencedArtifact,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
referencedArtifacts[referencedArtifact] = struct{}{}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
for varName, varValue := range artifactCfg.Vars {
|
for varName, varValue := range artifactCfg.Vars {
|
||||||
@@ -363,6 +409,17 @@ func validateScriptorium(cfg *ScriptoriumConfig) error {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
for artifactName := range referencedArtifacts {
|
||||||
|
artifactCfg := cfg.Artifacts[artifactName]
|
||||||
|
if strings.TrimSpace(artifactCfg.OutputPath) == "" {
|
||||||
|
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.output_path is required when artifact is referenced", artifactName)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if err := validateEnabledArtifactDependencyCycles(cfg.Artifacts); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -439,16 +496,41 @@ func archiveUploadConfiguredForS3(pipeline *PipelineConfig) bool {
|
|||||||
return enabled && upload
|
return enabled && upload
|
||||||
}
|
}
|
||||||
|
|
||||||
func isSupportedScriptoriumInputSource(source string) bool {
|
var windowsAbsPathRE = regexp.MustCompile(`^[A-Za-z]:[\\/].*`)
|
||||||
|
var envVarNameRE = regexp.MustCompile(`^[A-Za-z_][A-Za-z0-9_]*$`)
|
||||||
|
var scriptoriumArtifactKeyRE = regexp.MustCompile(`^[a-z][a-z0-9_]*$`)
|
||||||
|
var narratioArtifactSourceRE = regexp.MustCompile(`^narratio\.artifact\.([a-z][a-z0-9_]*)$`)
|
||||||
|
|
||||||
|
func validateScriptoriumInputSource(artifactName, inputName, source string, configuredArtifacts map[string]struct{}) (string, error) {
|
||||||
|
if isStaticSupportedScriptoriumInputSource(source) {
|
||||||
|
return "", nil
|
||||||
|
}
|
||||||
|
matches := narratioArtifactSourceRE.FindStringSubmatch(source)
|
||||||
|
if len(matches) != 2 {
|
||||||
|
return "", fmt.Errorf(
|
||||||
|
"pipeline.scriptorium.artifacts.%s.inputs.%s.source %q is unsupported",
|
||||||
|
artifactName,
|
||||||
|
inputName,
|
||||||
|
source,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
referenced := matches[1]
|
||||||
|
if _, ok := configuredArtifacts[referenced]; !ok {
|
||||||
|
return "", fmt.Errorf(
|
||||||
|
"pipeline.scriptorium.artifacts.%s.inputs.%s.source %q references unknown artifact %q",
|
||||||
|
artifactName,
|
||||||
|
inputName,
|
||||||
|
source,
|
||||||
|
referenced,
|
||||||
|
)
|
||||||
|
}
|
||||||
|
return referenced, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func isStaticSupportedScriptoriumInputSource(source string) bool {
|
||||||
switch strings.TrimSpace(source) {
|
switch strings.TrimSpace(source) {
|
||||||
case "previous_session_artifact":
|
case "previous_session_artifact":
|
||||||
return true
|
return true
|
||||||
case "processed_transcript":
|
|
||||||
return true
|
|
||||||
case "normalized_transcript":
|
|
||||||
return true
|
|
||||||
case "trimmed_transcript":
|
|
||||||
return true
|
|
||||||
case "narratio.transcript.merged":
|
case "narratio.transcript.merged":
|
||||||
return true
|
return true
|
||||||
case "narratio.transcript.polished":
|
case "narratio.transcript.polished":
|
||||||
@@ -459,16 +541,11 @@ func isSupportedScriptoriumInputSource(source string) bool {
|
|||||||
return true
|
return true
|
||||||
case "narratio.bounds.session":
|
case "narratio.bounds.session":
|
||||||
return true
|
return true
|
||||||
case "narratio.artifact.session_recap":
|
|
||||||
return true
|
|
||||||
default:
|
default:
|
||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
var windowsAbsPathRE = regexp.MustCompile(`^[A-Za-z]:[\\/].*`)
|
|
||||||
var envVarNameRE = regexp.MustCompile(`^[A-Za-z_][A-Za-z0-9_]*$`)
|
|
||||||
|
|
||||||
func validateEnvVarNameField(fieldName, value string) error {
|
func validateEnvVarNameField(fieldName, value string) error {
|
||||||
trimmed := strings.TrimSpace(value)
|
trimmed := strings.TrimSpace(value)
|
||||||
if trimmed == "" {
|
if trimmed == "" {
|
||||||
@@ -498,6 +575,73 @@ func validateRelativeSafePath(fieldName, value string) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func validatePathWithinRoot(fieldName, value, root string) error {
|
||||||
|
normalizedValue := filepath.ToSlash(filepath.Clean(strings.TrimSpace(value)))
|
||||||
|
normalizedRoot := filepath.ToSlash(filepath.Clean(strings.TrimSpace(root)))
|
||||||
|
if normalizedValue == normalizedRoot {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
if strings.HasPrefix(normalizedValue, normalizedRoot+"/") {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
return fmt.Errorf("%s must be under %s/", fieldName, normalizedRoot)
|
||||||
|
}
|
||||||
|
|
||||||
|
func validateEnabledArtifactDependencyCycles(artifacts map[string]ScriptoriumArtifactConfig) error {
|
||||||
|
if len(artifacts) == 0 {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
enabled := make(map[string]struct{}, len(artifacts))
|
||||||
|
graph := make(map[string][]string, len(artifacts))
|
||||||
|
for name, cfg := range artifacts {
|
||||||
|
if !cfg.Enabled {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
enabled[name] = struct{}{}
|
||||||
|
}
|
||||||
|
for name, cfg := range artifacts {
|
||||||
|
if !cfg.Enabled {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
for _, dep := range cfg.DependsOn {
|
||||||
|
trimmedDep := strings.TrimSpace(dep)
|
||||||
|
if _, ok := enabled[trimmedDep]; ok {
|
||||||
|
graph[name] = append(graph[name], trimmedDep)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
visiting := make(map[string]bool, len(enabled))
|
||||||
|
visited := make(map[string]bool, len(enabled))
|
||||||
|
|
||||||
|
var visit func(node string) error
|
||||||
|
visit = func(node string) error {
|
||||||
|
if visiting[node] {
|
||||||
|
return fmt.Errorf("pipeline.scriptorium.artifacts enabled dependencies must not contain cycles")
|
||||||
|
}
|
||||||
|
if visited[node] {
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
visiting[node] = true
|
||||||
|
for _, dep := range graph[node] {
|
||||||
|
if err := visit(dep); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
visiting[node] = false
|
||||||
|
visited[node] = true
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
for node := range enabled {
|
||||||
|
if err := visit(node); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
func validateDuration(fieldName, value string) error {
|
func validateDuration(fieldName, value string) error {
|
||||||
trimmed := strings.TrimSpace(value)
|
trimmed := strings.TrimSpace(value)
|
||||||
if trimmed == "" {
|
if trimmed == "" {
|
||||||
|
|||||||
@@ -28,6 +28,7 @@ type InputRecord struct {
|
|||||||
// ArtifactRecord captures one produced artifact and optional remote metadata.
|
// ArtifactRecord captures one produced artifact and optional remote metadata.
|
||||||
type ArtifactRecord struct {
|
type ArtifactRecord struct {
|
||||||
Kind string `json:"kind"`
|
Kind string `json:"kind"`
|
||||||
|
SourceID string `json:"source_id,omitempty"`
|
||||||
LocalPath string `json:"local_path"`
|
LocalPath string `json:"local_path"`
|
||||||
// ProducerRunID identifies the run that produced this durable artifact.
|
// ProducerRunID identifies the run that produced this durable artifact.
|
||||||
ProducerRunID string `json:"producer_run_id,omitempty"`
|
ProducerRunID string `json:"producer_run_id,omitempty"`
|
||||||
|
|||||||
@@ -28,12 +28,23 @@ func (analyzeStage) Declares() IODecl {
|
|||||||
{Kind: "transcript_normalized", Category: "transcripts", RelativePath: "transcripts/normalized.json"},
|
{Kind: "transcript_normalized", Category: "transcripts", RelativePath: "transcripts/normalized.json"},
|
||||||
{Kind: "transcript_trimmed", Category: "transcripts", RelativePath: "transcripts/trimmed.json"},
|
{Kind: "transcript_trimmed", Category: "transcripts", RelativePath: "transcripts/trimmed.json"},
|
||||||
},
|
},
|
||||||
Outputs: []artifacts.Ref{
|
Outputs: nil,
|
||||||
{Kind: "session_recap", Category: "artifacts", RelativePath: "artifacts/session_recap.md"},
|
|
||||||
},
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
type analyzeArtifactExecutionPlan struct {
|
||||||
|
Name string
|
||||||
|
Cfg config.ScriptoriumArtifactConfig
|
||||||
|
}
|
||||||
|
|
||||||
|
type analyzeArtifactExecutionResult struct {
|
||||||
|
Output artifacts.Ref
|
||||||
|
Logs []string
|
||||||
|
GeneratedConfigs []string
|
||||||
|
Metadata map[string]any
|
||||||
|
ReusedArtifacts []map[string]any
|
||||||
|
}
|
||||||
|
|
||||||
func (analyzeStage) Run(ctx context.Context, env *Env, m *manifest.Manifest) (*StageResult, error) {
|
func (analyzeStage) Run(ctx context.Context, env *Env, m *manifest.Manifest) (*StageResult, error) {
|
||||||
if env == nil || env.Config == nil {
|
if env == nil || env.Config == nil {
|
||||||
return nil, fmt.Errorf("analyze: stage environment config is required")
|
return nil, fmt.Errorf("analyze: stage environment config is required")
|
||||||
@@ -65,64 +76,286 @@ func (analyzeStage) Run(ctx context.Context, env *Env, m *manifest.Manifest) (*S
|
|||||||
return nil, fmt.Errorf("analyze: resolve run-stage layout: %w", err)
|
return nil, fmt.Errorf("analyze: resolve run-stage layout: %w", err)
|
||||||
}
|
}
|
||||||
if env.Config.Pipeline.Scriptorium == nil {
|
if env.Config.Pipeline.Scriptorium == nil {
|
||||||
return &StageResult{
|
return &StageResult{Metadata: map[string]any{
|
||||||
Metadata: map[string]any{
|
|
||||||
"stage": "analyze",
|
"stage": "analyze",
|
||||||
"skipped": true,
|
"skipped": true,
|
||||||
"reason": "pipeline.scriptorium is not configured",
|
"reason": "pipeline.scriptorium is not configured",
|
||||||
},
|
}}, nil
|
||||||
}, nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
artifactName, artifactCfg, skipReason, err := selectAnalyzeArtifact(env.Config.Pipeline.Scriptorium)
|
runtimeCatalog, err := buildAnalyzeRuntimeArtifactCatalog(paths, env.Config.Pipeline.Scriptorium, env.SelectedAnalyzeArtifacts)
|
||||||
|
if err != nil {
|
||||||
|
return nil, fmt.Errorf("analyze: build runtime artifact catalog: %w", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
plans, skipReason, err := buildAnalyzeExecutionPlans(env.Config.Pipeline.Scriptorium, runtimeCatalog)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("analyze: %w", err)
|
return nil, fmt.Errorf("analyze: %w", err)
|
||||||
}
|
}
|
||||||
if skipReason != "" {
|
if skipReason != "" {
|
||||||
return &StageResult{
|
return &StageResult{Metadata: map[string]any{
|
||||||
Metadata: map[string]any{
|
|
||||||
"stage": "analyze",
|
"stage": "analyze",
|
||||||
"skipped": true,
|
"skipped": true,
|
||||||
"reason": skipReason,
|
"reason": skipReason,
|
||||||
},
|
}}, nil
|
||||||
}, nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
transcriptRefs := discoverAnalyzeTranscriptRefs(m, paths)
|
transcriptRefs := discoverAnalyzeTranscriptRefs(m, paths)
|
||||||
|
sessionDir := filepath.Dir(strings.TrimSpace(env.Config.SessionPath))
|
||||||
|
|
||||||
|
outputs := make([]artifacts.Ref, 0, len(plans))
|
||||||
|
logs := []string{}
|
||||||
|
generatedConfigs := []string{}
|
||||||
|
artifactMetadata := make([]map[string]any, 0, len(plans))
|
||||||
|
reusedArtifacts := []map[string]any{}
|
||||||
|
reusedSeen := map[string]struct{}{}
|
||||||
|
|
||||||
|
for _, plan := range plans {
|
||||||
|
artifactResult, err := executeAnalyzeArtifact(
|
||||||
|
ctx,
|
||||||
|
env,
|
||||||
|
m,
|
||||||
|
paths,
|
||||||
|
runLayout,
|
||||||
|
sessionID,
|
||||||
|
sessionDir,
|
||||||
|
plan,
|
||||||
|
transcriptRefs,
|
||||||
|
runtimeCatalog,
|
||||||
|
)
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
outputs = append(outputs, artifactResult.Output)
|
||||||
|
logs = append(logs, artifactResult.Logs...)
|
||||||
|
generatedConfigs = append(generatedConfigs, artifactResult.GeneratedConfigs...)
|
||||||
|
artifactMetadata = append(artifactMetadata, artifactResult.Metadata)
|
||||||
|
for _, reused := range artifactResult.ReusedArtifacts {
|
||||||
|
sourceID, _ := reused["source_id"].(string)
|
||||||
|
path, _ := reused["path"].(string)
|
||||||
|
key := sourceID + "|" + path
|
||||||
|
if _, exists := reusedSeen[key]; exists {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
reusedSeen[key] = struct{}{}
|
||||||
|
reusedArtifacts = append(reusedArtifacts, reused)
|
||||||
|
}
|
||||||
|
|
||||||
|
sourceID, ok := runtimeCatalog.SourceIDForConfiguredKey(plan.Name)
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("analyze: source id not found for artifact %q", plan.Name)
|
||||||
|
}
|
||||||
|
if err := runtimeCatalog.MarkAvailableGenerated(sourceID, artifactResult.Output.AbsolutePath); err != nil {
|
||||||
|
return nil, fmt.Errorf("analyze: mark generated artifact %q available: %w", sourceID, err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
metadata := map[string]any{
|
||||||
|
"stage": "analyze",
|
||||||
|
"selected_artifacts": extractPlanNames(plans),
|
||||||
|
"generated_artifacts": artifactMetadata,
|
||||||
|
"reused_artifacts": reusedArtifacts,
|
||||||
|
"artifact_count": len(artifactMetadata),
|
||||||
|
"reused_artifact_count": len(reusedArtifacts),
|
||||||
|
}
|
||||||
|
if len(artifactMetadata) == 1 {
|
||||||
|
for k, v := range artifactMetadata[0] {
|
||||||
|
metadata[k] = v
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return &StageResult{
|
||||||
|
Outputs: outputs,
|
||||||
|
Logs: dedupeAndSortPaths(logs),
|
||||||
|
GeneratedConfigs: dedupeAndSortPaths(generatedConfigs),
|
||||||
|
Metadata: metadata,
|
||||||
|
}, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func buildAnalyzeExecutionPlans(
|
||||||
|
scriptoriumCfg *config.ScriptoriumConfig,
|
||||||
|
catalog *artifacts.ArtifactCatalog,
|
||||||
|
) ([]analyzeArtifactExecutionPlan, string, error) {
|
||||||
|
if scriptoriumCfg == nil {
|
||||||
|
return nil, "pipeline.scriptorium is not configured", nil
|
||||||
|
}
|
||||||
|
if len(scriptoriumCfg.Artifacts) == 0 {
|
||||||
|
return nil, "no scriptorium artifacts configured", nil
|
||||||
|
}
|
||||||
|
|
||||||
|
entries := catalog.ListConfigured()
|
||||||
|
selected := make([]string, 0, len(entries))
|
||||||
|
for _, entry := range entries {
|
||||||
|
if entry.Executable {
|
||||||
|
selected = append(selected, entry.ConfiguredKey)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if len(selected) == 0 {
|
||||||
|
return nil, "no selected scriptorium artifacts to execute", nil
|
||||||
|
}
|
||||||
|
|
||||||
|
ordered, err := orderSelectedScriptoriumArtifacts(scriptoriumCfg.Artifacts, selected, catalog)
|
||||||
|
if err != nil {
|
||||||
|
return nil, "", err
|
||||||
|
}
|
||||||
|
|
||||||
|
plans := make([]analyzeArtifactExecutionPlan, 0, len(ordered))
|
||||||
|
for _, name := range ordered {
|
||||||
|
artifactCfg, ok := scriptoriumCfg.Artifacts[name]
|
||||||
|
if !ok {
|
||||||
|
return nil, "", fmt.Errorf("selected artifact %q is not configured", name)
|
||||||
|
}
|
||||||
|
plans = append(plans, analyzeArtifactExecutionPlan{Name: name, Cfg: artifactCfg})
|
||||||
|
}
|
||||||
|
return plans, "", nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func orderSelectedScriptoriumArtifacts(
|
||||||
|
artifactsCfg map[string]config.ScriptoriumArtifactConfig,
|
||||||
|
selected []string,
|
||||||
|
catalog *artifacts.ArtifactCatalog,
|
||||||
|
) ([]string, error) {
|
||||||
|
selectedSet := map[string]struct{}{}
|
||||||
|
for _, key := range selected {
|
||||||
|
trimmed := strings.TrimSpace(key)
|
||||||
|
if trimmed == "" {
|
||||||
|
return nil, fmt.Errorf("selected artifact key must be non-empty")
|
||||||
|
}
|
||||||
|
selectedSet[trimmed] = struct{}{}
|
||||||
|
}
|
||||||
|
|
||||||
|
for selectedKey := range selectedSet {
|
||||||
|
cfg, ok := artifactsCfg[selectedKey]
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("selected artifact %q is not configured", selectedKey)
|
||||||
|
}
|
||||||
|
for _, dep := range cfg.DependsOn {
|
||||||
|
trimmedDep := strings.TrimSpace(dep)
|
||||||
|
if trimmedDep == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if _, ok := selectedSet[trimmedDep]; ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
sourceID, ok := catalog.SourceIDForConfiguredKey(trimmedDep)
|
||||||
|
if !ok {
|
||||||
|
return nil, fmt.Errorf("artifact %q depends on unknown configured artifact %q", selectedKey, trimmedDep)
|
||||||
|
}
|
||||||
|
entry, ok := catalog.Lookup(sourceID)
|
||||||
|
if !ok || !entry.Available {
|
||||||
|
return nil, fmt.Errorf("artifact %q depends on %q, but %q is unavailable", selectedKey, trimmedDep, sourceID)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
indegree := map[string]int{}
|
||||||
|
edges := map[string][]string{}
|
||||||
|
for key := range selectedSet {
|
||||||
|
indegree[key] = 0
|
||||||
|
}
|
||||||
|
for key := range selectedSet {
|
||||||
|
cfg := artifactsCfg[key]
|
||||||
|
for _, dep := range cfg.DependsOn {
|
||||||
|
trimmedDep := strings.TrimSpace(dep)
|
||||||
|
if _, ok := selectedSet[trimmedDep]; !ok {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
edges[trimmedDep] = append(edges[trimmedDep], key)
|
||||||
|
indegree[key]++
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
for key := range edges {
|
||||||
|
sort.Strings(edges[key])
|
||||||
|
}
|
||||||
|
|
||||||
|
ready := make([]string, 0, len(indegree))
|
||||||
|
for key, degree := range indegree {
|
||||||
|
if degree == 0 {
|
||||||
|
ready = append(ready, key)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
sort.Strings(ready)
|
||||||
|
|
||||||
|
order := make([]string, 0, len(selectedSet))
|
||||||
|
for len(ready) > 0 {
|
||||||
|
node := ready[0]
|
||||||
|
ready = ready[1:]
|
||||||
|
order = append(order, node)
|
||||||
|
for _, dep := range edges[node] {
|
||||||
|
indegree[dep]--
|
||||||
|
if indegree[dep] == 0 {
|
||||||
|
ready = append(ready, dep)
|
||||||
|
sort.Strings(ready)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if len(order) != len(selectedSet) {
|
||||||
|
return nil, fmt.Errorf("selected scriptorium artifacts contain a dependency cycle")
|
||||||
|
}
|
||||||
|
return order, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
func executeAnalyzeArtifact(
|
||||||
|
ctx context.Context,
|
||||||
|
env *Env,
|
||||||
|
m *manifest.Manifest,
|
||||||
|
paths artifacts.SessionPaths,
|
||||||
|
runLayout runStageLayout,
|
||||||
|
sessionID string,
|
||||||
|
sessionDir string,
|
||||||
|
plan analyzeArtifactExecutionPlan,
|
||||||
|
transcriptRefs analyzeTranscriptInputs,
|
||||||
|
runtimeCatalog *artifacts.ArtifactCatalog,
|
||||||
|
) (*analyzeArtifactExecutionResult, error) {
|
||||||
|
artifactName := plan.Name
|
||||||
|
artifactCfg := plan.Cfg
|
||||||
|
|
||||||
inputPaths := map[string]string{}
|
inputPaths := map[string]string{}
|
||||||
omittedOptionalInputs := []string{}
|
omittedOptionalInputs := []string{}
|
||||||
sessionDir := filepath.Dir(strings.TrimSpace(env.Config.SessionPath))
|
reusedArtifacts := []map[string]any{}
|
||||||
|
|
||||||
inputNames := sortedScriptoriumInputNames(artifactCfg.Inputs)
|
inputNames := sortedScriptoriumInputNames(artifactCfg.Inputs)
|
||||||
for _, inputName := range inputNames {
|
for _, inputName := range inputNames {
|
||||||
inputCfg := artifactCfg.Inputs[inputName]
|
inputCfg := artifactCfg.Inputs[inputName]
|
||||||
resolvedPath, resolved, resolveErr := resolveScriptoriumInput(inputName, inputCfg, m, paths, sessionDir)
|
resolvedPath, resolved, resolvedArtifact, resolveErr := resolveScriptoriumInput(inputName, inputCfg, m, paths, sessionDir, runtimeCatalog)
|
||||||
if resolveErr != nil {
|
if resolveErr != nil {
|
||||||
return nil, fmt.Errorf("analyze: resolve input %q: %w", inputName, resolveErr)
|
return nil, fmt.Errorf("analyze: resolve input %q for artifact %q: %w", inputName, artifactName, resolveErr)
|
||||||
}
|
}
|
||||||
if !resolved {
|
if !resolved {
|
||||||
if inputCfg.Required {
|
if inputCfg.Required {
|
||||||
return nil, fmt.Errorf("analyze: required input %q could not be resolved", inputName)
|
return nil, fmt.Errorf("analyze: required input %q for artifact %q could not be resolved", inputName, artifactName)
|
||||||
}
|
}
|
||||||
omittedOptionalInputs = append(omittedOptionalInputs, inputName)
|
omittedOptionalInputs = append(omittedOptionalInputs, inputName)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
inputPaths[inputName] = resolvedPath
|
inputPaths[inputName] = resolvedPath
|
||||||
|
if resolvedArtifact != nil && resolvedArtifact.Provenance == artifacts.ArtifactProvenanceDisabledFromDisk {
|
||||||
|
reusedArtifacts = append(reusedArtifacts, map[string]any{
|
||||||
|
"name": configuredArtifactNameFromSourceID(resolvedArtifact.ID),
|
||||||
|
"source_id": resolvedArtifact.ID,
|
||||||
|
"path": resolvedArtifact.Path,
|
||||||
|
"provenance": resolvedArtifact.Provenance,
|
||||||
|
})
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
vars, err := buildScriptoriumVars(artifactCfg.Vars, env.Config.Session)
|
vars, err := buildScriptoriumVars(artifactCfg.Vars, env.Config.Session)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("analyze: resolve vars: %w", err)
|
return nil, fmt.Errorf("analyze: resolve vars for artifact %q: %w", artifactName, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
canonicalOutputPath, err := resolveScriptoriumOutputPath(paths, artifactCfg.OutputPath)
|
canonicalOutputPath, err := resolveScriptoriumOutputPath(paths, artifactCfg.OutputPath)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("analyze: resolve output path: %w", err)
|
return nil, fmt.Errorf("analyze: resolve output path for artifact %q: %w", artifactName, err)
|
||||||
}
|
}
|
||||||
outputPath, err := runLocalPathForCanonical(runLayout, paths, canonicalOutputPath)
|
outputPath, err := runLocalPathForCanonical(runLayout, paths, canonicalOutputPath)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("analyze: resolve run-local output path: %w", err)
|
return nil, fmt.Errorf("analyze: resolve run-local output path for artifact %q: %w", artifactName, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
stdoutLogPath := filepath.Join(paths.LogsDir, "scriptorium."+artifactName+".stdout.log")
|
stdoutLogPath := filepath.Join(paths.LogsDir, "scriptorium."+artifactName+".stdout.log")
|
||||||
stderrLogPath := filepath.Join(paths.LogsDir, "scriptorium."+artifactName+".stderr.log")
|
stderrLogPath := filepath.Join(paths.LogsDir, "scriptorium."+artifactName+".stderr.log")
|
||||||
generatedConfigPath := filepath.Join(paths.ConfigDir, "scriptorium."+artifactName+".generated.yml")
|
generatedConfigPath := filepath.Join(paths.ConfigDir, "scriptorium."+artifactName+".generated.yml")
|
||||||
@@ -134,14 +367,17 @@ func (analyzeStage) Run(ctx context.Context, env *Env, m *manifest.Manifest) (*S
|
|||||||
|
|
||||||
timeout, err := resolveScriptoriumTimeout(env.Config.Pipeline.Scriptorium.Timeout, artifactCfg.Timeout)
|
timeout, err := resolveScriptoriumTimeout(env.Config.Pipeline.Scriptorium.Timeout, artifactCfg.Timeout)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("analyze: resolve timeout: %w", err)
|
return nil, fmt.Errorf("analyze: resolve timeout for artifact %q: %w", artifactName, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
logPaths := []string{}
|
logPaths := []string{}
|
||||||
generatedConfigs := []string{}
|
generatedConfigs := []string{}
|
||||||
meta := map[string]any{
|
meta := map[string]any{
|
||||||
"stage": "analyze",
|
"stage": "analyze",
|
||||||
|
"name": artifactName,
|
||||||
"artifact_name": artifactName,
|
"artifact_name": artifactName,
|
||||||
|
"source_id": artifacts.ConfiguredArtifactSourceID(artifactName),
|
||||||
|
"output_kind": "scriptorium_artifact",
|
||||||
"prompt_id": artifactCfg.PromptID,
|
"prompt_id": artifactCfg.PromptID,
|
||||||
"profile_id": artifactCfg.ProfileID,
|
"profile_id": artifactCfg.ProfileID,
|
||||||
"binary": env.Config.Pipeline.Scriptorium.Binary,
|
"binary": env.Config.Pipeline.Scriptorium.Binary,
|
||||||
@@ -189,10 +425,10 @@ func (analyzeStage) Run(ctx context.Context, env *Env, m *manifest.Manifest) (*S
|
|||||||
}
|
}
|
||||||
renderRes, renderErr := env.Scriptorium.RenderArtifact(ctx, renderReq)
|
renderRes, renderErr := env.Scriptorium.RenderArtifact(ctx, renderReq)
|
||||||
if renderErr != nil {
|
if renderErr != nil {
|
||||||
return nil, fmt.Errorf("analyze: scriptorium render failed: %w", renderErr)
|
return nil, fmt.Errorf("analyze: scriptorium render failed for artifact %q: %w", artifactName, renderErr)
|
||||||
}
|
}
|
||||||
if renderRes.ValidationFailed {
|
if renderRes.ValidationFailed {
|
||||||
return nil, fmt.Errorf("analyze: scriptorium render returned validation_failed=true")
|
return nil, fmt.Errorf("analyze: scriptorium render returned validation_failed=true for artifact %q", artifactName)
|
||||||
}
|
}
|
||||||
finalRenderOutputPath := coalesceString(renderRes.OutputPath, renderReq.OutputPath)
|
finalRenderOutputPath := coalesceString(renderRes.OutputPath, renderReq.OutputPath)
|
||||||
if err := requireNonEmptyFile(finalRenderOutputPath, artifactName+" render output"); err != nil {
|
if err := requireNonEmptyFile(finalRenderOutputPath, artifactName+" render output"); err != nil {
|
||||||
@@ -240,7 +476,8 @@ func (analyzeStage) Run(ctx context.Context, env *Env, m *manifest.Manifest) (*S
|
|||||||
if runErr != nil {
|
if runErr != nil {
|
||||||
if res.ValidationFailed {
|
if res.ValidationFailed {
|
||||||
return nil, fmt.Errorf(
|
return nil, fmt.Errorf(
|
||||||
"analyze: scriptorium validation failed (prompt_id=%q, output_path=%q, exit_code=%d, stdout_log=%q, stderr_log=%q): %w",
|
"analyze: scriptorium validation failed (artifact=%q, prompt_id=%q, output_path=%q, exit_code=%d, stdout_log=%q, stderr_log=%q): %w",
|
||||||
|
artifactName,
|
||||||
req.PromptID,
|
req.PromptID,
|
||||||
coalesceString(res.OutputPath, req.OutputPath),
|
coalesceString(res.OutputPath, req.OutputPath),
|
||||||
res.ExitCode,
|
res.ExitCode,
|
||||||
@@ -249,10 +486,10 @@ func (analyzeStage) Run(ctx context.Context, env *Env, m *manifest.Manifest) (*S
|
|||||||
runErr,
|
runErr,
|
||||||
)
|
)
|
||||||
}
|
}
|
||||||
return nil, fmt.Errorf("analyze: scriptorium run failed: %w", runErr)
|
return nil, fmt.Errorf("analyze: scriptorium run failed for artifact %q: %w", artifactName, runErr)
|
||||||
}
|
}
|
||||||
if res.ValidationFailed {
|
if res.ValidationFailed {
|
||||||
return nil, fmt.Errorf("analyze: scriptorium run returned validation_failed=true")
|
return nil, fmt.Errorf("analyze: scriptorium run returned validation_failed=true for artifact %q", artifactName)
|
||||||
}
|
}
|
||||||
|
|
||||||
finalOutputPath := coalesceString(res.OutputPath, req.OutputPath)
|
finalOutputPath := coalesceString(res.OutputPath, req.OutputPath)
|
||||||
@@ -265,12 +502,13 @@ func (analyzeStage) Run(ctx context.Context, env *Env, m *manifest.Manifest) (*S
|
|||||||
SessionID: sessionID,
|
SessionID: sessionID,
|
||||||
})
|
})
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("analyze: promote artifact output: %w", err)
|
return nil, fmt.Errorf("analyze: promote artifact output for %q: %w", artifactName, err)
|
||||||
}
|
}
|
||||||
|
|
||||||
logPaths = append(logPaths, stdoutLogPath, stderrLogPath)
|
logPaths = append(logPaths, stdoutLogPath, stderrLogPath)
|
||||||
generatedConfigs = append(generatedConfigs, generatedConfigPath)
|
generatedConfigs = append(generatedConfigs, generatedConfigPath)
|
||||||
meta["run_output_path"] = finalOutputPath
|
meta["run_output_path"] = finalOutputPath
|
||||||
|
meta["path"] = canonicalOutputPath
|
||||||
meta["output_path"] = canonicalOutputPath
|
meta["output_path"] = canonicalOutputPath
|
||||||
meta["generated_config_path"] = generatedConfigPath
|
meta["generated_config_path"] = generatedConfigPath
|
||||||
meta["stdout_log_path"] = stdoutLogPath
|
meta["stdout_log_path"] = stdoutLogPath
|
||||||
@@ -285,39 +523,38 @@ func (analyzeStage) Run(ctx context.Context, env *Env, m *manifest.Manifest) (*S
|
|||||||
meta["adapter_generated_config"] = res.GeneratedConfigPath
|
meta["adapter_generated_config"] = res.GeneratedConfigPath
|
||||||
meta["adapter_stdout_log_path"] = res.StdoutLogPath
|
meta["adapter_stdout_log_path"] = res.StdoutLogPath
|
||||||
meta["adapter_stderr_log_path"] = res.StderrLogPath
|
meta["adapter_stderr_log_path"] = res.StderrLogPath
|
||||||
|
meta["provenance"] = artifacts.ArtifactProvenanceGeneratedCurrentAnalyzeRun
|
||||||
if res.Metadata != nil {
|
if res.Metadata != nil {
|
||||||
meta["adapter_metadata"] = res.Metadata
|
meta["adapter_metadata"] = res.Metadata
|
||||||
}
|
}
|
||||||
|
|
||||||
return &StageResult{
|
return &analyzeArtifactExecutionResult{
|
||||||
Outputs: []artifacts.Ref{promotedArtifact},
|
Output: promotedArtifact,
|
||||||
Logs: logPaths,
|
Logs: logPaths,
|
||||||
GeneratedConfigs: generatedConfigs,
|
GeneratedConfigs: generatedConfigs,
|
||||||
Metadata: meta,
|
Metadata: meta,
|
||||||
|
ReusedArtifacts: reusedArtifacts,
|
||||||
}, nil
|
}, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
func selectAnalyzeArtifact(cfg *config.ScriptoriumConfig) (string, config.ScriptoriumArtifactConfig, string, error) {
|
func extractPlanNames(plans []analyzeArtifactExecutionPlan) []string {
|
||||||
if cfg == nil {
|
if len(plans) == 0 {
|
||||||
return "", config.ScriptoriumArtifactConfig{}, "pipeline.scriptorium is not configured", nil
|
return nil
|
||||||
|
}
|
||||||
|
out := make([]string, 0, len(plans))
|
||||||
|
for _, plan := range plans {
|
||||||
|
out = append(out, plan.Name)
|
||||||
|
}
|
||||||
|
return out
|
||||||
}
|
}
|
||||||
|
|
||||||
enabled := []string{}
|
func configuredArtifactNameFromSourceID(sourceID string) string {
|
||||||
for name, artifact := range cfg.Artifacts {
|
trimmed := strings.TrimSpace(sourceID)
|
||||||
if artifact.Enabled {
|
const prefix = "narratio.artifact."
|
||||||
enabled = append(enabled, name)
|
if !strings.HasPrefix(trimmed, prefix) {
|
||||||
|
return ""
|
||||||
}
|
}
|
||||||
}
|
return strings.TrimPrefix(trimmed, prefix)
|
||||||
sort.Strings(enabled)
|
|
||||||
if len(enabled) == 0 {
|
|
||||||
return "", config.ScriptoriumArtifactConfig{}, "no enabled scriptorium artifacts configured", nil
|
|
||||||
}
|
|
||||||
|
|
||||||
sessionRecapCfg, ok := cfg.Artifacts["session_recap"]
|
|
||||||
if !ok || !sessionRecapCfg.Enabled {
|
|
||||||
return "", config.ScriptoriumArtifactConfig{}, "", fmt.Errorf("only artifacts.session_recap is supported in this analyze implementation; enabled=%s", strings.Join(enabled, ","))
|
|
||||||
}
|
|
||||||
return "session_recap", sessionRecapCfg, "", nil
|
|
||||||
}
|
}
|
||||||
|
|
||||||
func discoverProcessedTranscript(m *manifest.Manifest, paths artifacts.SessionPaths) (string, string, error) {
|
func discoverProcessedTranscript(m *manifest.Manifest, paths artifacts.SessionPaths) (string, string, error) {
|
||||||
@@ -391,42 +628,97 @@ func resolveScriptoriumInput(
|
|||||||
m *manifest.Manifest,
|
m *manifest.Manifest,
|
||||||
paths artifacts.SessionPaths,
|
paths artifacts.SessionPaths,
|
||||||
sessionDir string,
|
sessionDir string,
|
||||||
) (string, bool, error) {
|
runtimeCatalog *artifacts.ArtifactCatalog,
|
||||||
switch strings.TrimSpace(inputCfg.Source) {
|
) (string, bool, *artifacts.ResolvedSessionArtifact, error) {
|
||||||
|
source := strings.TrimSpace(inputCfg.Source)
|
||||||
|
switch source {
|
||||||
case "previous_session_artifact":
|
case "previous_session_artifact":
|
||||||
if strings.TrimSpace(inputCfg.Path) == "" {
|
if strings.TrimSpace(inputCfg.Path) == "" {
|
||||||
return "", false, nil
|
return "", false, nil, nil
|
||||||
}
|
}
|
||||||
resolved := resolveInputPathForRead(paths, sessionDir, inputCfg.Path)
|
resolved := resolveInputPathForRead(paths, sessionDir, inputCfg.Path)
|
||||||
if err := requireFile(resolved, "scriptorium input "+inputName); err != nil {
|
if err := requireFile(resolved, "scriptorium input "+inputName); err != nil {
|
||||||
return "", false, nil
|
return "", false, nil, nil
|
||||||
}
|
}
|
||||||
return resolved, true, nil
|
return resolved, true, nil, nil
|
||||||
default:
|
default:
|
||||||
resolved, err := artifacts.ResolveSessionArtifact(paths, m, inputCfg.Source)
|
resolved, err := artifacts.ResolveSessionArtifactWithCatalog(paths, m, source, runtimeCatalog)
|
||||||
if err == nil {
|
if err == nil {
|
||||||
return resolved.Path, true, nil
|
copy := resolved
|
||||||
|
return resolved.Path, true, ©, nil
|
||||||
}
|
}
|
||||||
if errors.Is(err, artifacts.ErrSessionArtifactNotFound) {
|
if errors.Is(err, artifacts.ErrSessionArtifactNotFound) {
|
||||||
normalized, normalizeErr := artifacts.NormalizeSessionArtifactSource(inputCfg.Source)
|
if artifacts.IsConfiguredArtifactSource(source) {
|
||||||
|
if inputCfg.Required {
|
||||||
|
return "", false, nil, fmt.Errorf("configured artifact source %q is unavailable", source)
|
||||||
|
}
|
||||||
|
return "", false, nil, nil
|
||||||
|
}
|
||||||
|
normalized, normalizeErr := artifacts.NormalizeSessionArtifactSource(source)
|
||||||
if normalizeErr != nil {
|
if normalizeErr != nil {
|
||||||
return "", false, normalizeErr
|
return "", false, nil, normalizeErr
|
||||||
}
|
}
|
||||||
switch normalized {
|
switch normalized {
|
||||||
case artifacts.ArtifactTranscriptPolished:
|
case artifacts.ArtifactTranscriptPolished:
|
||||||
return "", false, nil
|
return "", false, nil, nil
|
||||||
case artifacts.ArtifactTranscriptFull:
|
case artifacts.ArtifactTranscriptFull:
|
||||||
return "", false, fmt.Errorf("normalized transcript input is unavailable; run normalize stage first")
|
return "", false, nil, fmt.Errorf("normalized transcript input is unavailable; run normalize stage first")
|
||||||
case artifacts.ArtifactTranscriptTrimmed:
|
case artifacts.ArtifactTranscriptTrimmed:
|
||||||
return "", false, fmt.Errorf("trimmed transcript input is unavailable; run trim stage first")
|
return "", false, nil, fmt.Errorf("trimmed transcript input is unavailable; run trim stage first")
|
||||||
default:
|
default:
|
||||||
return "", false, nil
|
return "", false, nil, nil
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
return "", false, err
|
return "", false, nil, err
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func buildAnalyzeRuntimeArtifactCatalog(
|
||||||
|
paths artifacts.SessionPaths,
|
||||||
|
scriptoriumCfg *config.ScriptoriumConfig,
|
||||||
|
selectedArtifacts []string,
|
||||||
|
) (*artifacts.ArtifactCatalog, error) {
|
||||||
|
catalog := artifacts.NewArtifactCatalog()
|
||||||
|
if err := catalog.RegisterBuiltIns(); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if scriptoriumCfg == nil {
|
||||||
|
return catalog, nil
|
||||||
|
}
|
||||||
|
|
||||||
|
configured := map[string]artifacts.ConfiguredArtifactDefinition{}
|
||||||
|
for key, artifactCfg := range scriptoriumCfg.Artifacts {
|
||||||
|
configured[key] = artifacts.ConfiguredArtifactDefinition{
|
||||||
|
Enabled: artifactCfg.Enabled,
|
||||||
|
OutputPath: artifactCfg.OutputPath,
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if err := catalog.RegisterConfiguredArtifacts(configured, selectedArtifacts); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
|
||||||
|
for _, entry := range catalog.ListConfigured() {
|
||||||
|
if entry.Executable {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if strings.TrimSpace(entry.CanonicalRelPath) == "" {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
resolvedPath, err := resolveScriptoriumOutputPath(paths, entry.CanonicalRelPath)
|
||||||
|
if err != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if err := requireNonEmptyFile(resolvedPath, "configured artifact "+entry.SourceID); err != nil {
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
if err := catalog.MarkAvailableFromDisk(entry.SourceID, resolvedPath); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
return catalog, nil
|
||||||
|
}
|
||||||
|
|
||||||
func resolveInputPathForRead(paths artifacts.SessionPaths, sessionDir, pathValue string) string {
|
func resolveInputPathForRead(paths artifacts.SessionPaths, sessionDir, pathValue string) string {
|
||||||
trimmed := strings.TrimSpace(pathValue)
|
trimmed := strings.TrimSpace(pathValue)
|
||||||
if trimmed == "" {
|
if trimmed == "" {
|
||||||
|
|||||||
@@ -231,7 +231,7 @@ func TestAnalyzeOmitsOptionalPreviousRecapWhenUnavailable(t *testing.T) {
|
|||||||
Timeout: "2m",
|
Timeout: "2m",
|
||||||
Inputs: map[string]config.ScriptoriumInputConfig{
|
Inputs: map[string]config.ScriptoriumInputConfig{
|
||||||
"transcript": {
|
"transcript": {
|
||||||
Source: "processed_transcript",
|
Source: "narratio.transcript.polished",
|
||||||
Required: true,
|
Required: true,
|
||||||
},
|
},
|
||||||
"previous_recap": {
|
"previous_recap": {
|
||||||
@@ -358,7 +358,7 @@ func TestAnalyzeIncludesPreviousRecapWhenConfiguredAndAvailable(t *testing.T) {
|
|||||||
Timeout: "2m",
|
Timeout: "2m",
|
||||||
Inputs: map[string]config.ScriptoriumInputConfig{
|
Inputs: map[string]config.ScriptoriumInputConfig{
|
||||||
"transcript": {
|
"transcript": {
|
||||||
Source: "processed_transcript",
|
Source: "narratio.transcript.polished",
|
||||||
Required: true,
|
Required: true,
|
||||||
},
|
},
|
||||||
"previous_recap": {
|
"previous_recap": {
|
||||||
@@ -394,7 +394,7 @@ func TestAnalyzeFailsWhenRequiredPreviousRecapMissing(t *testing.T) {
|
|||||||
OutputPath: "artifacts/session_recap.md",
|
OutputPath: "artifacts/session_recap.md",
|
||||||
Inputs: map[string]config.ScriptoriumInputConfig{
|
Inputs: map[string]config.ScriptoriumInputConfig{
|
||||||
"transcript": {
|
"transcript": {
|
||||||
Source: "processed_transcript",
|
Source: "narratio.transcript.polished",
|
||||||
Required: true,
|
Required: true,
|
||||||
},
|
},
|
||||||
"previous_recap": {
|
"previous_recap": {
|
||||||
@@ -414,6 +414,309 @@ func TestAnalyzeFailsWhenRequiredPreviousRecapMissing(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestAnalyzeResolvesConfiguredArtifactInputFromDisabledArtifactOutput(t *testing.T) {
|
||||||
|
env, m, fake := setupAnalyzeEnv(t)
|
||||||
|
paths := sessionPathsForEnv(env, m.SessionID)
|
||||||
|
writeAnalyzeFile(t, filepath.Join(paths.TranscriptsDir, "trimmed.json"), `{"segments":[]}`)
|
||||||
|
playerHandoutPath := filepath.Join(paths.ArtifactsDir, "player_handout.md")
|
||||||
|
writeAnalyzeFile(t, playerHandoutPath, "handout\n")
|
||||||
|
|
||||||
|
sessionRecap := env.Config.Pipeline.Scriptorium.Artifacts["session_recap"]
|
||||||
|
sessionRecap.Inputs["recap"] = config.ScriptoriumInputConfig{
|
||||||
|
Source: "narratio.artifact.player_handout",
|
||||||
|
Required: true,
|
||||||
|
}
|
||||||
|
env.Config.Pipeline.Scriptorium.Artifacts["session_recap"] = sessionRecap
|
||||||
|
env.Config.Pipeline.Scriptorium.Artifacts["player_handout"] = config.ScriptoriumArtifactConfig{
|
||||||
|
Enabled: false,
|
||||||
|
OutputPath: "artifacts/player_handout.md",
|
||||||
|
}
|
||||||
|
|
||||||
|
_, err := (analyzeStage{}).Run(context.Background(), env, m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v", err)
|
||||||
|
}
|
||||||
|
if len(fake.RunRequests) != 1 {
|
||||||
|
t.Fatalf("run requests = %d, want 1", len(fake.RunRequests))
|
||||||
|
}
|
||||||
|
if got := fake.RunRequests[0].InputPaths["recap"]; got != playerHandoutPath {
|
||||||
|
t.Fatalf("recap input = %q, want %q", got, playerHandoutPath)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAnalyzeMetadataIncludesGeneratedAndReusedArtifacts(t *testing.T) {
|
||||||
|
env, m, _ := setupAnalyzeEnv(t)
|
||||||
|
paths := sessionPathsForEnv(env, m.SessionID)
|
||||||
|
writeAnalyzeFile(t, filepath.Join(paths.TranscriptsDir, "trimmed.json"), `{"segments":[]}`)
|
||||||
|
playerHandoutPath := filepath.Join(paths.ArtifactsDir, "player_handout.md")
|
||||||
|
writeAnalyzeFile(t, playerHandoutPath, "handout\n")
|
||||||
|
|
||||||
|
sessionRecap := env.Config.Pipeline.Scriptorium.Artifacts["session_recap"]
|
||||||
|
sessionRecap.Inputs["recap"] = config.ScriptoriumInputConfig{
|
||||||
|
Source: "narratio.artifact.player_handout",
|
||||||
|
Required: true,
|
||||||
|
}
|
||||||
|
env.Config.Pipeline.Scriptorium.Artifacts["session_recap"] = sessionRecap
|
||||||
|
env.Config.Pipeline.Scriptorium.Artifacts["player_handout"] = config.ScriptoriumArtifactConfig{
|
||||||
|
Enabled: false,
|
||||||
|
OutputPath: "artifacts/player_handout.md",
|
||||||
|
}
|
||||||
|
|
||||||
|
result, err := (analyzeStage{}).Run(context.Background(), env, m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v", err)
|
||||||
|
}
|
||||||
|
|
||||||
|
generated := mustArtifactEntryList(t, result.Metadata, "generated_artifacts")
|
||||||
|
if len(generated) != 1 {
|
||||||
|
t.Fatalf("generated_artifacts len = %d, want 1 (%#v)", len(generated), generated)
|
||||||
|
}
|
||||||
|
g0 := generated[0]
|
||||||
|
if g0["name"] != "session_recap" {
|
||||||
|
t.Fatalf("generated[0].name = %#v, want session_recap", g0["name"])
|
||||||
|
}
|
||||||
|
if g0["source_id"] != "narratio.artifact.session_recap" {
|
||||||
|
t.Fatalf("generated[0].source_id = %#v, want narratio.artifact.session_recap", g0["source_id"])
|
||||||
|
}
|
||||||
|
if g0["output_kind"] != "scriptorium_artifact" {
|
||||||
|
t.Fatalf("generated[0].output_kind = %#v, want scriptorium_artifact", g0["output_kind"])
|
||||||
|
}
|
||||||
|
if g0["path"] != filepath.Join(paths.ArtifactsDir, "session_recap.md") {
|
||||||
|
t.Fatalf("generated[0].path = %#v, want session recap path", g0["path"])
|
||||||
|
}
|
||||||
|
if g0["prompt_id"] != "dnd.session_recap" {
|
||||||
|
t.Fatalf("generated[0].prompt_id = %#v, want dnd.session_recap", g0["prompt_id"])
|
||||||
|
}
|
||||||
|
if g0["profile_id"] != "local-quality" {
|
||||||
|
t.Fatalf("generated[0].profile_id = %#v, want local-quality", g0["profile_id"])
|
||||||
|
}
|
||||||
|
if g0["provenance"] != artifacts.ArtifactProvenanceGeneratedCurrentAnalyzeRun {
|
||||||
|
t.Fatalf("generated[0].provenance = %#v, want %q", g0["provenance"], artifacts.ArtifactProvenanceGeneratedCurrentAnalyzeRun)
|
||||||
|
}
|
||||||
|
|
||||||
|
reused := mustArtifactEntryList(t, result.Metadata, "reused_artifacts")
|
||||||
|
if len(reused) != 1 {
|
||||||
|
t.Fatalf("reused_artifacts len = %d, want 1 (%#v)", len(reused), reused)
|
||||||
|
}
|
||||||
|
r0 := reused[0]
|
||||||
|
if r0["name"] != "player_handout" {
|
||||||
|
t.Fatalf("reused[0].name = %#v, want player_handout", r0["name"])
|
||||||
|
}
|
||||||
|
if r0["source_id"] != "narratio.artifact.player_handout" {
|
||||||
|
t.Fatalf("reused[0].source_id = %#v, want narratio.artifact.player_handout", r0["source_id"])
|
||||||
|
}
|
||||||
|
if r0["path"] != playerHandoutPath {
|
||||||
|
t.Fatalf("reused[0].path = %#v, want %q", r0["path"], playerHandoutPath)
|
||||||
|
}
|
||||||
|
if r0["provenance"] != artifacts.ArtifactProvenanceDisabledFromDisk {
|
||||||
|
t.Fatalf("reused[0].provenance = %#v, want %q", r0["provenance"], artifacts.ArtifactProvenanceDisabledFromDisk)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAnalyzeRunsMultipleIndependentArtifactsInDeterministicOrder(t *testing.T) {
|
||||||
|
env, m, fake := setupAnalyzeEnv(t)
|
||||||
|
paths := sessionPathsForEnv(env, m.SessionID)
|
||||||
|
writeAnalyzeFile(t, filepath.Join(paths.TranscriptsDir, "trimmed.json"), `{"segments":[]}`)
|
||||||
|
|
||||||
|
env.Config.Pipeline.Scriptorium.Artifacts["player_handout"] = config.ScriptoriumArtifactConfig{
|
||||||
|
Enabled: true,
|
||||||
|
PromptID: "dnd.player_handout",
|
||||||
|
ProfileID: "local-quality",
|
||||||
|
OutputPath: "artifacts/player_handout.md",
|
||||||
|
Inputs: map[string]config.ScriptoriumInputConfig{
|
||||||
|
"transcript": {
|
||||||
|
Source: "narratio.transcript.trimmed",
|
||||||
|
Required: true,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
result, err := (analyzeStage{}).Run(context.Background(), env, m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v", err)
|
||||||
|
}
|
||||||
|
if len(fake.RunRequests) != 2 {
|
||||||
|
t.Fatalf("run requests = %d, want 2", len(fake.RunRequests))
|
||||||
|
}
|
||||||
|
if fake.RunRequests[0].PromptID != "dnd.player_handout" {
|
||||||
|
t.Fatalf("first prompt id = %q, want dnd.player_handout", fake.RunRequests[0].PromptID)
|
||||||
|
}
|
||||||
|
if fake.RunRequests[1].PromptID != "dnd.session_recap" {
|
||||||
|
t.Fatalf("second prompt id = %q, want dnd.session_recap", fake.RunRequests[1].PromptID)
|
||||||
|
}
|
||||||
|
|
||||||
|
selected, ok := result.Metadata["selected_artifacts"].([]string)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("selected_artifacts = %#v, want []string", result.Metadata["selected_artifacts"])
|
||||||
|
}
|
||||||
|
if len(selected) != 2 || selected[0] != "player_handout" || selected[1] != "session_recap" {
|
||||||
|
t.Fatalf("selected_artifacts = %#v, want [player_handout session_recap]", selected)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAnalyzeRunsDependenciesBeforeDependents(t *testing.T) {
|
||||||
|
env, m, fake := setupAnalyzeEnv(t)
|
||||||
|
paths := sessionPathsForEnv(env, m.SessionID)
|
||||||
|
writeAnalyzeFile(t, filepath.Join(paths.TranscriptsDir, "trimmed.json"), `{"segments":[]}`)
|
||||||
|
|
||||||
|
env.Config.Pipeline.Scriptorium.Artifacts["player_handout"] = config.ScriptoriumArtifactConfig{
|
||||||
|
Enabled: true,
|
||||||
|
DependsOn: []string{"session_recap"},
|
||||||
|
PromptID: "dnd.player_handout",
|
||||||
|
ProfileID: "local-quality",
|
||||||
|
OutputPath: "artifacts/player_handout.md",
|
||||||
|
Inputs: map[string]config.ScriptoriumInputConfig{
|
||||||
|
"recap": {
|
||||||
|
Source: "narratio.artifact.session_recap",
|
||||||
|
Required: true,
|
||||||
|
},
|
||||||
|
"transcript": {
|
||||||
|
Source: "narratio.transcript.trimmed",
|
||||||
|
Required: true,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
_, err := (analyzeStage{}).Run(context.Background(), env, m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v", err)
|
||||||
|
}
|
||||||
|
if len(fake.RunRequests) != 2 {
|
||||||
|
t.Fatalf("run requests = %d, want 2", len(fake.RunRequests))
|
||||||
|
}
|
||||||
|
if fake.RunRequests[0].PromptID != "dnd.session_recap" {
|
||||||
|
t.Fatalf("first prompt id = %q, want dnd.session_recap", fake.RunRequests[0].PromptID)
|
||||||
|
}
|
||||||
|
if fake.RunRequests[1].PromptID != "dnd.player_handout" {
|
||||||
|
t.Fatalf("second prompt id = %q, want dnd.player_handout", fake.RunRequests[1].PromptID)
|
||||||
|
}
|
||||||
|
if got := fake.RunRequests[1].InputPaths["recap"]; got != filepath.Join(paths.ArtifactsDir, "session_recap.md") {
|
||||||
|
t.Fatalf("dependent recap path = %q, want %q", got, filepath.Join(paths.ArtifactsDir, "session_recap.md"))
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAnalyzeAppliesSelectedArtifactsFilter(t *testing.T) {
|
||||||
|
env, m, fake := setupAnalyzeEnv(t)
|
||||||
|
paths := sessionPathsForEnv(env, m.SessionID)
|
||||||
|
writeAnalyzeFile(t, filepath.Join(paths.TranscriptsDir, "trimmed.json"), `{"segments":[]}`)
|
||||||
|
|
||||||
|
env.Config.Pipeline.Scriptorium.Artifacts["player_handout"] = config.ScriptoriumArtifactConfig{
|
||||||
|
Enabled: true,
|
||||||
|
PromptID: "dnd.player_handout",
|
||||||
|
ProfileID: "local-quality",
|
||||||
|
OutputPath: "artifacts/player_handout.md",
|
||||||
|
Inputs: map[string]config.ScriptoriumInputConfig{
|
||||||
|
"transcript": {
|
||||||
|
Source: "narratio.transcript.trimmed",
|
||||||
|
Required: true,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
env.SelectedAnalyzeArtifacts = []string{"player_handout"}
|
||||||
|
|
||||||
|
result, err := (analyzeStage{}).Run(context.Background(), env, m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v", err)
|
||||||
|
}
|
||||||
|
if len(fake.RunRequests) != 1 {
|
||||||
|
t.Fatalf("run requests = %d, want 1", len(fake.RunRequests))
|
||||||
|
}
|
||||||
|
if fake.RunRequests[0].PromptID != "dnd.player_handout" {
|
||||||
|
t.Fatalf("prompt id = %q, want dnd.player_handout", fake.RunRequests[0].PromptID)
|
||||||
|
}
|
||||||
|
if len(result.Outputs) != 1 || result.Outputs[0].Kind != "player_handout" {
|
||||||
|
t.Fatalf("outputs = %#v, want only player_handout", result.Outputs)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAnalyzeMetadataIncludesMultipleGeneratedArtifacts(t *testing.T) {
|
||||||
|
env, m, _ := setupAnalyzeEnv(t)
|
||||||
|
paths := sessionPathsForEnv(env, m.SessionID)
|
||||||
|
writeAnalyzeFile(t, filepath.Join(paths.TranscriptsDir, "trimmed.json"), `{"segments":[]}`)
|
||||||
|
|
||||||
|
env.Config.Pipeline.Scriptorium.Artifacts["player_handout"] = config.ScriptoriumArtifactConfig{
|
||||||
|
Enabled: true,
|
||||||
|
PromptID: "dnd.player_handout",
|
||||||
|
ProfileID: "local-quality",
|
||||||
|
OutputPath: "artifacts/player_handout.md",
|
||||||
|
Inputs: map[string]config.ScriptoriumInputConfig{
|
||||||
|
"transcript": {
|
||||||
|
Source: "narratio.transcript.trimmed",
|
||||||
|
Required: true,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
result, err := (analyzeStage{}).Run(context.Background(), env, m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v", err)
|
||||||
|
}
|
||||||
|
generated := mustArtifactEntryList(t, result.Metadata, "generated_artifacts")
|
||||||
|
if len(generated) != 2 {
|
||||||
|
t.Fatalf("generated_artifacts len = %d, want 2 (%#v)", len(generated), generated)
|
||||||
|
}
|
||||||
|
for i, entry := range generated {
|
||||||
|
for _, field := range []string{"name", "source_id", "output_kind", "path", "prompt_id", "profile_id", "provenance"} {
|
||||||
|
if _, ok := entry[field]; !ok {
|
||||||
|
t.Fatalf("generated[%d] missing field %q: %#v", i, field, entry)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAnalyzeFailsWhenRequiredConfiguredArtifactMissing(t *testing.T) {
|
||||||
|
env, m, _ := setupAnalyzeEnv(t)
|
||||||
|
paths := sessionPathsForEnv(env, m.SessionID)
|
||||||
|
writeAnalyzeFile(t, filepath.Join(paths.TranscriptsDir, "trimmed.json"), `{"segments":[]}`)
|
||||||
|
|
||||||
|
sessionRecap := env.Config.Pipeline.Scriptorium.Artifacts["session_recap"]
|
||||||
|
sessionRecap.Inputs["recap"] = config.ScriptoriumInputConfig{
|
||||||
|
Source: "narratio.artifact.player_handout",
|
||||||
|
Required: true,
|
||||||
|
}
|
||||||
|
env.Config.Pipeline.Scriptorium.Artifacts["session_recap"] = sessionRecap
|
||||||
|
env.Config.Pipeline.Scriptorium.Artifacts["player_handout"] = config.ScriptoriumArtifactConfig{
|
||||||
|
Enabled: false,
|
||||||
|
OutputPath: "artifacts/player_handout.md",
|
||||||
|
}
|
||||||
|
|
||||||
|
_, err := (analyzeStage{}).Run(context.Background(), env, m)
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error, got nil")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), `configured artifact source "narratio.artifact.player_handout" is unavailable`) {
|
||||||
|
t.Fatalf("error = %q, want configured artifact unavailable context", err.Error())
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func TestAnalyzeOmitsOptionalMissingConfiguredArtifactInput(t *testing.T) {
|
||||||
|
env, m, fake := setupAnalyzeEnv(t)
|
||||||
|
paths := sessionPathsForEnv(env, m.SessionID)
|
||||||
|
writeAnalyzeFile(t, filepath.Join(paths.TranscriptsDir, "trimmed.json"), `{"segments":[]}`)
|
||||||
|
|
||||||
|
sessionRecap := env.Config.Pipeline.Scriptorium.Artifacts["session_recap"]
|
||||||
|
sessionRecap.Inputs["recap"] = config.ScriptoriumInputConfig{
|
||||||
|
Source: "narratio.artifact.player_handout",
|
||||||
|
Required: false,
|
||||||
|
}
|
||||||
|
env.Config.Pipeline.Scriptorium.Artifacts["session_recap"] = sessionRecap
|
||||||
|
env.Config.Pipeline.Scriptorium.Artifacts["player_handout"] = config.ScriptoriumArtifactConfig{
|
||||||
|
Enabled: false,
|
||||||
|
OutputPath: "artifacts/player_handout.md",
|
||||||
|
}
|
||||||
|
|
||||||
|
_, err := (analyzeStage{}).Run(context.Background(), env, m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v", err)
|
||||||
|
}
|
||||||
|
if len(fake.RunRequests) != 1 {
|
||||||
|
t.Fatalf("run requests = %d, want 1", len(fake.RunRequests))
|
||||||
|
}
|
||||||
|
if _, exists := fake.RunRequests[0].InputPaths["recap"]; exists {
|
||||||
|
t.Fatalf("optional recap input should be omitted when unavailable, got %q", fake.RunRequests[0].InputPaths["recap"])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func TestAnalyzeFailsWhenOutputPathMissing(t *testing.T) {
|
func TestAnalyzeFailsWhenOutputPathMissing(t *testing.T) {
|
||||||
env, m, fake := setupAnalyzeEnv(t)
|
env, m, fake := setupAnalyzeEnv(t)
|
||||||
paths := sessionPathsForEnv(env, m.SessionID)
|
paths := sessionPathsForEnv(env, m.SessionID)
|
||||||
@@ -456,7 +759,7 @@ func TestAnalyzeSupportsProcessedTranscriptSourceWhenConfigured(t *testing.T) {
|
|||||||
|
|
||||||
artifact := env.Config.Pipeline.Scriptorium.Artifacts["session_recap"]
|
artifact := env.Config.Pipeline.Scriptorium.Artifacts["session_recap"]
|
||||||
artifact.Inputs["transcript"] = config.ScriptoriumInputConfig{
|
artifact.Inputs["transcript"] = config.ScriptoriumInputConfig{
|
||||||
Source: "processed_transcript",
|
Source: "narratio.transcript.polished",
|
||||||
Required: true,
|
Required: true,
|
||||||
}
|
}
|
||||||
env.Config.Pipeline.Scriptorium.Artifacts["session_recap"] = artifact
|
env.Config.Pipeline.Scriptorium.Artifacts["session_recap"] = artifact
|
||||||
@@ -505,7 +808,7 @@ func TestAnalyzeSupportsNormalizedTranscriptSourceWhenConfigured(t *testing.T) {
|
|||||||
|
|
||||||
artifact := env.Config.Pipeline.Scriptorium.Artifacts["session_recap"]
|
artifact := env.Config.Pipeline.Scriptorium.Artifacts["session_recap"]
|
||||||
artifact.Inputs["transcript"] = config.ScriptoriumInputConfig{
|
artifact.Inputs["transcript"] = config.ScriptoriumInputConfig{
|
||||||
Source: "normalized_transcript",
|
Source: "narratio.transcript.full",
|
||||||
Required: true,
|
Required: true,
|
||||||
}
|
}
|
||||||
env.Config.Pipeline.Scriptorium.Artifacts["session_recap"] = artifact
|
env.Config.Pipeline.Scriptorium.Artifacts["session_recap"] = artifact
|
||||||
@@ -565,7 +868,7 @@ func TestAnalyzeSupportsNormalizedTranscriptSourceFromManifestOutput(t *testing.
|
|||||||
|
|
||||||
artifact := env.Config.Pipeline.Scriptorium.Artifacts["session_recap"]
|
artifact := env.Config.Pipeline.Scriptorium.Artifacts["session_recap"]
|
||||||
artifact.Inputs["transcript"] = config.ScriptoriumInputConfig{
|
artifact.Inputs["transcript"] = config.ScriptoriumInputConfig{
|
||||||
Source: "normalized_transcript",
|
Source: "narratio.transcript.full",
|
||||||
Required: true,
|
Required: true,
|
||||||
}
|
}
|
||||||
env.Config.Pipeline.Scriptorium.Artifacts["session_recap"] = artifact
|
env.Config.Pipeline.Scriptorium.Artifacts["session_recap"] = artifact
|
||||||
@@ -586,7 +889,7 @@ func TestAnalyzeFailsWhenNormalizedTranscriptMissing(t *testing.T) {
|
|||||||
env, m, fake := setupAnalyzeEnv(t)
|
env, m, fake := setupAnalyzeEnv(t)
|
||||||
artifact := env.Config.Pipeline.Scriptorium.Artifacts["session_recap"]
|
artifact := env.Config.Pipeline.Scriptorium.Artifacts["session_recap"]
|
||||||
artifact.Inputs["transcript"] = config.ScriptoriumInputConfig{
|
artifact.Inputs["transcript"] = config.ScriptoriumInputConfig{
|
||||||
Source: "normalized_transcript",
|
Source: "narratio.transcript.full",
|
||||||
Required: true,
|
Required: true,
|
||||||
}
|
}
|
||||||
env.Config.Pipeline.Scriptorium.Artifacts["session_recap"] = artifact
|
env.Config.Pipeline.Scriptorium.Artifacts["session_recap"] = artifact
|
||||||
@@ -715,6 +1018,22 @@ func TestAnalyzeSkipsWhenNoEnabledScriptoriumArtifactsConfigured(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestAnalyzeSkipsWhenArtifactMapEmpty(t *testing.T) {
|
||||||
|
env, m, _ := setupAnalyzeEnv(t)
|
||||||
|
env.Config.Pipeline.Scriptorium.Artifacts = map[string]config.ScriptoriumArtifactConfig{}
|
||||||
|
|
||||||
|
result, err := (analyzeStage{}).Run(context.Background(), env, m)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("Run() error = %v", err)
|
||||||
|
}
|
||||||
|
if result.Metadata["skipped"] != true {
|
||||||
|
t.Fatalf("metadata = %#v, want skipped=true", result.Metadata)
|
||||||
|
}
|
||||||
|
if result.Metadata["reason"] != "no scriptorium artifacts configured" {
|
||||||
|
t.Fatalf("reason = %#v, want no scriptorium artifacts configured", result.Metadata["reason"])
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
func setupAnalyzeEnv(t *testing.T) (*Env, *manifest.Manifest, *scriptorium.FakeRunner) {
|
func setupAnalyzeEnv(t *testing.T) (*Env, *manifest.Manifest, *scriptorium.FakeRunner) {
|
||||||
t.Helper()
|
t.Helper()
|
||||||
workspace := t.TempDir()
|
workspace := t.TempDir()
|
||||||
@@ -744,7 +1063,7 @@ func setupAnalyzeEnv(t *testing.T) (*Env, *manifest.Manifest, *scriptorium.FakeR
|
|||||||
Timeout: "2m",
|
Timeout: "2m",
|
||||||
Inputs: map[string]config.ScriptoriumInputConfig{
|
Inputs: map[string]config.ScriptoriumInputConfig{
|
||||||
"transcript": {
|
"transcript": {
|
||||||
Source: "trimmed_transcript",
|
Source: "narratio.transcript.trimmed",
|
||||||
Required: true,
|
Required: true,
|
||||||
},
|
},
|
||||||
"previous_recap": {
|
"previous_recap": {
|
||||||
@@ -802,3 +1121,27 @@ func writeAnalyzeFileNoTest(path, contents string) {
|
|||||||
_ = os.MkdirAll(filepath.Dir(path), 0o755)
|
_ = os.MkdirAll(filepath.Dir(path), 0o755)
|
||||||
_ = os.WriteFile(path, []byte(contents), 0o644)
|
_ = os.WriteFile(path, []byte(contents), 0o644)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func mustArtifactEntryList(t *testing.T, metadata map[string]any, key string) []map[string]any {
|
||||||
|
t.Helper()
|
||||||
|
raw, ok := metadata[key]
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("metadata missing key %q: %#v", key, metadata)
|
||||||
|
}
|
||||||
|
if typed, ok := raw.([]map[string]any); ok {
|
||||||
|
return typed
|
||||||
|
}
|
||||||
|
asList, ok := raw.([]any)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("metadata[%q] = %#v, want []map[string]any", key, raw)
|
||||||
|
}
|
||||||
|
out := make([]map[string]any, 0, len(asList))
|
||||||
|
for _, item := range asList {
|
||||||
|
m, ok := item.(map[string]any)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("metadata[%q] entry = %#v, want map[string]any", key, item)
|
||||||
|
}
|
||||||
|
out = append(out, m)
|
||||||
|
}
|
||||||
|
return out
|
||||||
|
}
|
||||||
|
|||||||
@@ -19,6 +19,7 @@ import (
|
|||||||
// Env is the shared dependency container visible to stages.
|
// Env is the shared dependency container visible to stages.
|
||||||
type Env struct {
|
type Env struct {
|
||||||
Config *config.Config
|
Config *config.Config
|
||||||
|
SelectedAnalyzeArtifacts []string
|
||||||
ArtifactStore artifacts.Store
|
ArtifactStore artifacts.Store
|
||||||
ManifestStore manifest.Store
|
ManifestStore manifest.Store
|
||||||
Logger *slog.Logger
|
Logger *slog.Logger
|
||||||
|
|||||||
Reference in New Issue
Block a user