Compare commits
77 Commits
13de820931
...
v1.5.0
| Author | SHA1 | Date | |
|---|---|---|---|
| 8b4b328c4e | |||
| ee2b8e63e6 | |||
| 51edd384c0 | |||
| b804d0f2c8 | |||
| fd5ccc668b | |||
| 0dc8ff9b52 | |||
| 8657a28bdb | |||
| a9c5e4ad4e | |||
| 2176b4371d | |||
| effc10d75b | |||
| 5887839aa1 | |||
| 3128bef20a | |||
| 4e4e2b7d96 | |||
| 99f4f9a0db | |||
| 6abdd67bb5 | |||
| c32e0c401f | |||
| ab5751459a | |||
| 62de6abdbf | |||
| 903dc70682 | |||
| 23c714da66 | |||
| 6639775d7d | |||
| 966b95b176 | |||
| 3bcf2c08dd | |||
| 700ab655ca | |||
| 85c5647385 | |||
| 2ef7c76d99 | |||
| 9bc1b0feda | |||
| 5cec84a4a7 | |||
| abfbe42d61 | |||
| e7e3bef1e4 | |||
| a68e8e31a4 | |||
| 3a9e60cda9 | |||
| 905ff03ccc | |||
| 495f7bcde4 | |||
| 51e0e8c5d0 | |||
| a2409a1fd1 | |||
| b3363f87d6 | |||
| 5831c0c9e6 | |||
| 42ed81cbe1 | |||
| e433c86203 | |||
| a2a144dffa | |||
| 8ef6e99d69 | |||
| 2545faef6c | |||
| 80be8be4d6 | |||
| 801adb385d | |||
| feba7b9d74 | |||
| b89224bbde | |||
| 131ffd9887 | |||
| af492c9e97 | |||
| f39fc94610 | |||
| 4e4eff6ba7 | |||
| 8ff1b4fa66 | |||
| 702f622e18 | |||
| 9da2c1e144 | |||
| 32653f54f9 | |||
| 72a200968a | |||
| b39b68add7 | |||
| d9fa1d9328 | |||
| 8375ad83f3 | |||
| 4158394dcf | |||
| eac7e155a5 | |||
| 0cf2cbfeb3 | |||
| 361dbb4ca8 | |||
| d6deccf3e8 | |||
| ee747243fe | |||
| a1ceb457e9 | |||
| 9900211fa4 | |||
| 60cebf0e4b | |||
| 7bd575187e | |||
| ab5a7e8e3d | |||
| 99b2e1cd81 | |||
| 363313d99c | |||
| 18ddf00d3d | |||
| 59f3fe3d1d | |||
| 1dccf5f140 | |||
| 0b40cf8026 | |||
| a7ec195587 |
@@ -2,8 +2,33 @@ when:
|
||||
- event: tag
|
||||
|
||||
steps:
|
||||
- name: build-release-assets
|
||||
validate:
|
||||
image: golang:1.25
|
||||
commands:
|
||||
- go test ./...
|
||||
- go test -race ./...
|
||||
- go vet ./...
|
||||
- go build ./...
|
||||
- go test ./internal/doccheck
|
||||
- go test ./internal/config -run '^TestExamplesLoadAndValidate$'
|
||||
|
||||
cross-build:
|
||||
image: golang:1.25
|
||||
depends_on: validate
|
||||
commands:
|
||||
- |
|
||||
set -eu
|
||||
output_dir="$(mktemp -d)"
|
||||
trap 'rm -rf "$output_dir"' EXIT
|
||||
for target in linux/amd64 linux/arm64 darwin/amd64 darwin/arm64 windows/amd64 windows/arm64; do
|
||||
goos="${target%/*}"
|
||||
goarch="${target#*/}"
|
||||
CGO_ENABLED=0 GOOS="$goos" GOARCH="$goarch" go build -o "$output_dir/narratio-$goos-$goarch" ./cmd/narratio
|
||||
done
|
||||
|
||||
build-release-assets:
|
||||
image: golang:1.25
|
||||
depends_on: [validate, cross-build]
|
||||
commands:
|
||||
- |
|
||||
set -eu
|
||||
@@ -33,7 +58,17 @@ steps:
|
||||
build_binary windows amd64 ".exe"
|
||||
build_binary windows arm64 ".exe"
|
||||
|
||||
- name: publish-release
|
||||
smoke_binary="$dist/narratio-version-smoke"
|
||||
go build -trimpath -ldflags "-s -w -X gitea.maximumdirect.net/eric/narratio/internal/buildinfo.Version=$version" \
|
||||
-o "$smoke_binary" "$pkg"
|
||||
reported_version="$("$smoke_binary" version)"
|
||||
rm -f "$smoke_binary"
|
||||
if [ "$reported_version" != "narratio $version" ]; then
|
||||
echo "release binary reported unexpected version: $reported_version" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
publish-release:
|
||||
image: woodpeckerci/plugin-release
|
||||
depends_on:
|
||||
- build-release-assets
|
||||
|
||||
8
.woodpecker/shuffle.yml
Normal file
8
.woodpecker/shuffle.yml
Normal file
@@ -0,0 +1,8 @@
|
||||
when:
|
||||
- event: cron
|
||||
|
||||
steps:
|
||||
shuffled-race-tests:
|
||||
image: golang:1.25
|
||||
commands:
|
||||
- go test -race -shuffle=on -count=3 ./...
|
||||
47
.woodpecker/verify.yml
Normal file
47
.woodpecker/verify.yml
Normal file
@@ -0,0 +1,47 @@
|
||||
when:
|
||||
- event: [push, pull_request]
|
||||
|
||||
steps:
|
||||
tests:
|
||||
image: golang:1.25
|
||||
commands:
|
||||
- go test ./...
|
||||
|
||||
race-tests:
|
||||
image: golang:1.25
|
||||
depends_on: tests
|
||||
commands:
|
||||
- go test -race ./...
|
||||
|
||||
static-analysis:
|
||||
image: golang:1.25
|
||||
depends_on: tests
|
||||
commands:
|
||||
- go vet ./...
|
||||
|
||||
build:
|
||||
image: golang:1.25
|
||||
depends_on: tests
|
||||
commands:
|
||||
- go build ./...
|
||||
|
||||
documentation-and-examples:
|
||||
image: golang:1.25
|
||||
depends_on: tests
|
||||
commands:
|
||||
- go test ./internal/doccheck
|
||||
- go test ./internal/config -run '^TestExamplesLoadAndValidate$'
|
||||
|
||||
cross-build:
|
||||
image: golang:1.25
|
||||
depends_on: [race-tests, static-analysis, build, documentation-and-examples]
|
||||
commands:
|
||||
- |
|
||||
set -eu
|
||||
output_dir="$(mktemp -d)"
|
||||
trap 'rm -rf "$output_dir"' EXIT
|
||||
for target in linux/amd64 linux/arm64 darwin/amd64 darwin/arm64 windows/amd64 windows/arm64; do
|
||||
goos="${target%/*}"
|
||||
goarch="${target#*/}"
|
||||
CGO_ENABLED=0 GOOS="$goos" GOARCH="$goarch" go build -o "$output_dir/narratio-$goos-$goarch" ./cmd/narratio
|
||||
done
|
||||
89
docs/cli.md
89
docs/cli.md
@@ -12,7 +12,9 @@ This runs the canonical full pipeline for session `2026-04-04`.
|
||||
|
||||
Top-level commands:
|
||||
|
||||
- `run <session_id>`: run full stage order.
|
||||
- `version`: print the Narratio build version.
|
||||
- `run <session_id>`: run all or one contiguous range of the canonical stage order.
|
||||
- `regenerate-artifacts <session_id>`: force-run extraction through analysis.
|
||||
- `run-stage <stage> <session_id>`: run one stage.
|
||||
- `analyze <session_id>`: force-run analyze.
|
||||
- `publish <session_id>`: force-run publish.
|
||||
@@ -47,7 +49,11 @@ Rules:
|
||||
- `--campaign` and `--campaign-file` are mutually exclusive.
|
||||
- `--session` is not used by `session init`.
|
||||
- if both positional `<session_id>` and `--session-id` are provided, values must match.
|
||||
- `--previous-session-id` is a strict expectation: the selected session file
|
||||
must contain the same `previous_session_id`.
|
||||
- `clean --all` cannot be combined with campaign/session selectors.
|
||||
- notification delivery is currently limited to the configured `noop` mode; see
|
||||
the [configuration reference](./config.md#notifications).
|
||||
|
||||
## Session ID Input Rules
|
||||
|
||||
@@ -66,22 +72,63 @@ Commands with additional positionals keep their command-specific order:
|
||||
|
||||
## Command Reference
|
||||
|
||||
### `version`
|
||||
|
||||
```bash
|
||||
narratio version
|
||||
```
|
||||
|
||||
Official release binaries report their exact Git tag. Binaries built directly
|
||||
from source without release linker metadata report `dev`.
|
||||
|
||||
### `run`
|
||||
|
||||
```bash
|
||||
narratio run <session_id> [--force] [--artifacts <name[,name...]>] [...common config flags]
|
||||
narratio run <session_id> [--from <stage>] [--through <stage>] [--force] [--artifacts <name[,name...]>] [...common config flags]
|
||||
```
|
||||
|
||||
Behavior:
|
||||
|
||||
- evaluates full stage order;
|
||||
- runs `extract` between `trim` and `render`; an omitted or disabled Notarius
|
||||
- evaluates one inclusive contiguous range of the canonical stage order;
|
||||
- defaults an omitted `--from` to `prepare` and an omitted `--through` to
|
||||
`notify`, so omitting both retains full-pipeline behavior;
|
||||
- rejects unknown endpoints and a `--from` endpoint after `--through`;
|
||||
- runs `render` before `extract`; an omitted or disabled Notarius
|
||||
configuration records an explicit `notarius_disabled` self-skip;
|
||||
- skips already-succeeded stages unless `--force` is set or a stage-specific
|
||||
resume check finds its durable result obsolete;
|
||||
- applies `--force` only to stages in the selected range;
|
||||
- rejects repeated `--from`, `--through`, or `--force` options, including
|
||||
`--name=value` spellings;
|
||||
- continues interrupted or partially completed sessions by running non-succeeded stages;
|
||||
- writes session and run manifests.
|
||||
|
||||
When `--artifacts` is present, the selected range must contain `analyze` or
|
||||
`publish`. Either consumer is sufficient, including a one-stage range.
|
||||
|
||||
### `regenerate-artifacts`
|
||||
|
||||
```bash
|
||||
narratio regenerate-artifacts <session_id> [--artifacts <name[,name...]>] [...common config flags]
|
||||
```
|
||||
|
||||
Exactly equivalent to:
|
||||
|
||||
```bash
|
||||
narratio run <session_id> --force --from extract --through analyze [caller options]
|
||||
```
|
||||
|
||||
The command always reruns extraction. Analysis rebuilds the selected configured
|
||||
artifacts and any prerequisites required by those targets; without
|
||||
`--artifacts`, it uses the normal default analysis selection. Publish and notify
|
||||
never run. Common session/configuration options and repeatable artifact values
|
||||
pass through unchanged.
|
||||
|
||||
Because the expansion owns `--force`, `--from`, and `--through`, callers cannot
|
||||
supply those options. The shared `run` parser reports them as duplicate
|
||||
singleton flags. The alias has no private execution options or behavior, and
|
||||
runtime diagnostics may identify the operation as `run`.
|
||||
|
||||
### `run-stage`
|
||||
|
||||
```bash
|
||||
@@ -96,8 +143,8 @@ Valid stage names:
|
||||
- `polish`
|
||||
- `normalize`
|
||||
- `trim`
|
||||
- `extract`
|
||||
- `render`
|
||||
- `extract`
|
||||
- `analyze`
|
||||
- `publish`
|
||||
- `notify`
|
||||
@@ -149,10 +196,17 @@ post-publish cleanup behavior.
|
||||
### `session plan`
|
||||
|
||||
```bash
|
||||
narratio session plan <session_id> [--force] [...common config flags]
|
||||
narratio session plan <session_id> [--from <stage>] [--through <stage>] [--force] [--artifacts <name[,name...]>] [...common config flags]
|
||||
```
|
||||
|
||||
Validates config, prepares local workdir layout, and prints run/skip decisions for each stage.
|
||||
Uses the same inclusive bounds, endpoint validation, force scope, and artifact
|
||||
selection contract as `run`. It validates config and prints run/skip decisions
|
||||
for selected stages only without creating the local workdir or changing the
|
||||
manifest. Resume-capable selected stages are checked against durable evidence.
|
||||
For `analyze`, the preview also lists explicit targets, prerequisite-only work,
|
||||
execution order, and reusable current artifacts with concise reasons. These
|
||||
artifact decisions come from the same reconciliation and work planner used by
|
||||
execution; the preview does not predict output identities.
|
||||
|
||||
### `session validate`
|
||||
|
||||
@@ -168,7 +222,8 @@ Read-only preflight checks for config validity, required inputs, audio mode, pre
|
||||
narratio session status <session_id> [...common config flags]
|
||||
```
|
||||
|
||||
Prints local manifest state and, when storage is available, remote current-state and published-output status.
|
||||
Prints local manifest state and, when storage is available, status for the
|
||||
pointer-selected remote commit and its declared published outputs.
|
||||
|
||||
### `session init`
|
||||
|
||||
@@ -211,7 +266,8 @@ Behavior:
|
||||
- discovers committed remote current state;
|
||||
- plans local restores;
|
||||
- writes an execution report;
|
||||
- blocks conflicting overwrites unless `--force` is set.
|
||||
- blocks unresolved conflicts. `--force` permits replacement only of eligible
|
||||
regular files.
|
||||
|
||||
See [Operations: Restore Workflow](./operations.md#restore-workflow) for the
|
||||
default restore scope, report location, and conflict-handling workflow.
|
||||
@@ -246,14 +302,17 @@ and precedence.
|
||||
|
||||
## `--artifacts` Selection Rules
|
||||
|
||||
- accepted on `run`, `run-stage`, `analyze`, and `publish`;
|
||||
- accepted on `run`, `session plan`, `run-stage`, `analyze`, and `publish`;
|
||||
- repeatable and comma-separated values are combined, surrounding whitespace
|
||||
is removed, and duplicate names are collapsed;
|
||||
- names must exist in `pipeline.scriptorium.artifacts`;
|
||||
- empty entries are invalid;
|
||||
- repeated names are deduplicated.
|
||||
- on `run-stage`, only `analyze` and `publish` accept the option.
|
||||
|
||||
Effects:
|
||||
|
||||
- filters analyze execution to selected configured artifacts;
|
||||
- selects explicit analyze targets; required configured prerequisites may be
|
||||
reused or rebuilt before them;
|
||||
- filters publish rules that source `narratio.artifact.<name>`;
|
||||
- does not filter built-in transcript/bounds or explicitly configured
|
||||
`narratio.extraction.<name>` publish sources; and
|
||||
@@ -285,6 +344,12 @@ Force publish only:
|
||||
narratio publish 2026-04-04
|
||||
```
|
||||
|
||||
Regenerate post-transcript artifacts without publishing:
|
||||
|
||||
```bash
|
||||
narratio regenerate-artifacts 2026-04-04 --artifacts session_recap,player_handout
|
||||
```
|
||||
|
||||
## Output And Exit Behavior
|
||||
|
||||
- Successful commands write their result or summary to standard output and
|
||||
|
||||
130
docs/config.md
130
docs/config.md
@@ -38,13 +38,32 @@ If local session discovery fails and a `session_id` is known, Narratio attempts
|
||||
|
||||
using configured object storage.
|
||||
|
||||
The downloaded remote session file is command-scoped: Narratio removes it after
|
||||
the command finishes and records only the remote object provenance alongside
|
||||
the durable copied session input.
|
||||
|
||||
### Identity segments
|
||||
|
||||
Campaign IDs (`campaign_id` and `default_campaign_id`), session IDs, previous
|
||||
session IDs, and Narratio run IDs are opaque portable segments. They must use
|
||||
only ASCII letters, digits, `.`, `_`, and `-`; empty values, `.`/`..`, path
|
||||
separators, drive forms, whitespace, control characters, and non-ASCII text are
|
||||
rejected. Narratio does not trim or rewrite these values. Existing manifests or
|
||||
remote state with an unsafe legacy identity must be migrated before use.
|
||||
|
||||
## Validation and Merge Rules
|
||||
|
||||
- YAML decode is strict (`KnownFields(true)`): unknown fields fail load.
|
||||
- YAML decode is strict (`KnownFields(true)`) and accepts exactly one document:
|
||||
unknown fields or trailing documents fail load.
|
||||
- Configured timeout and retry-delay durations must be positive. An omitted
|
||||
artifact timeout continues to inherit its configured Scriptorium timeout.
|
||||
- Session files must be concrete; unresolved `{{ ... }}` placeholders fail load.
|
||||
- Pipeline defaults are applied before validation.
|
||||
- Campaign and session identities must agree.
|
||||
- Stable files (`speakers_file`, `autocorrect_file`, `glossary_file`, `players_file`, `party_file`) resolve from session overrides when provided, otherwise from campaign defaults.
|
||||
- Required stable files (`speakers_file`, `autocorrect_file`, `glossary_file`,
|
||||
`players_file`, `party_file`) and the optional `spell_catalog_file` resolve
|
||||
from session overrides when provided, otherwise from campaign defaults. An
|
||||
empty or omitted session spell-catalog value inherits the campaign value.
|
||||
- Exactly one audio mode must be configured in session input:
|
||||
- local (`audio_dir` or `audio_files`), or
|
||||
- S3 (`audio_s3.prefix`).
|
||||
@@ -85,7 +104,13 @@ inputs:
|
||||
|
||||
- Do not place raw secrets in YAML.
|
||||
- Use env var names in config (for example `pipeline.audita.llm_api_key_env`).
|
||||
- Optionally load env files from `pipeline.secrets.env_dir`.
|
||||
- Optionally load credential files from `pipeline.secrets.env_dir`. Each valid
|
||||
environment-variable filename supplies one value; trailing CR/LF is removed.
|
||||
- An existing process environment value takes precedence over a credential file.
|
||||
- Credential directories and files must not be symlinks and must be regular,
|
||||
bounded files (at most 8 KiB per value). On POSIX, provision the directory
|
||||
with no group/other access (normally `0700`) and files with no group/other
|
||||
access (normally `0600`).
|
||||
- Commands that need storage/auth load filesystem secrets before constructing adapters.
|
||||
|
||||
## Publish Configuration Summary
|
||||
@@ -135,8 +160,8 @@ Rules:
|
||||
| `pipeline.campaigns.root` | string | No | `/usr/local/share/narratio/campaigns` |
|
||||
| `pipeline.campaigns.default_campaign_id` | string | No | empty |
|
||||
| `pipeline.secrets.env_dir` | string | No | empty |
|
||||
| `pipeline.storage.backend` | string | No | empty |
|
||||
| `pipeline.storage.s3.bucket` | string | Conditional | required for S3 session-audio and for publish upload when backend is `s3` |
|
||||
| `pipeline.storage.backend` | string | No | `local`; supported values are `local` and `s3` (case-insensitive) |
|
||||
| `pipeline.storage.s3.bucket` | string | Conditional | required when backend is `s3` and S3 session-audio or publish upload is enabled |
|
||||
| `pipeline.storage.s3.root_prefix` | string | No | `dnd` |
|
||||
| `pipeline.storage.s3.region` | string | No | empty |
|
||||
| `pipeline.storage.s3.endpoint` | string | No | empty |
|
||||
@@ -156,7 +181,7 @@ Rules:
|
||||
| `pipeline.publish.locks[]` | list | No | empty |
|
||||
| `pipeline.publish.locks[].source` | string | Yes (per lock) | must reference supported publish source |
|
||||
| `pipeline.publish.locks[].reason` | string | No | empty |
|
||||
| `pipeline.whisperx.transcribe_url` | string | Yes | valid URL |
|
||||
| `pipeline.whisperx.transcribe_url` | string | Yes | absolute `http` or `https` URL |
|
||||
| `pipeline.whisperx.language` | string | No | `en` |
|
||||
| `pipeline.whisperx.timeout` | duration | No | `30m` |
|
||||
| `pipeline.whisperx.retries` | int | No | `3` |
|
||||
@@ -205,6 +230,7 @@ Rules:
|
||||
| `pipeline.notarius.pipeline_id` | string | Conditional | required when enabled |
|
||||
| `pipeline.notarius.timeout` | duration | No | `3h`; must be positive |
|
||||
| `pipeline.notarius.working_directory` | string | No | directory containing resolved `config_path`; relative paths resolve from the pipeline file directory |
|
||||
| `pipeline.notarius.references` | map[string]string | No | empty; maps normalized Notarius selectors to supported prepared Narratio source IDs; maximum 256 entries |
|
||||
| `pipeline.notarius.outputs` | map | Conditional | at least one entry when enabled |
|
||||
| `pipeline.render.enabled` | bool | No | `true` |
|
||||
| `pipeline.render.format` | string | No | `markdown` (only supported value) |
|
||||
@@ -217,9 +243,45 @@ Rules:
|
||||
| `pipeline.scriptorium.timeout` | duration | No | `10m` |
|
||||
| `pipeline.scriptorium.render_debug` | bool | No | `false` |
|
||||
| `pipeline.scriptorium.artifacts` | map | No | empty |
|
||||
| `pipeline.notification.backend` | string | No | empty |
|
||||
| `pipeline.notification.recipient` | string | No | empty |
|
||||
| `pipeline.notification.timeout` | duration | No | empty |
|
||||
| `pipeline.notification.mode` | string | No | `noop`; the only supported notification mode until a provider is implemented |
|
||||
|
||||
### Notarius Reference Bindings
|
||||
|
||||
`pipeline.notarius.references` maps a Notarius CLI selector to a prepared
|
||||
Narratio source, not to a filesystem path:
|
||||
|
||||
```yaml
|
||||
notarius:
|
||||
references:
|
||||
glossary: narratio.input.glossary
|
||||
party: narratio.input.party
|
||||
players: narratio.input.players
|
||||
spell_catalog: narratio.input.spell_catalog
|
||||
```
|
||||
|
||||
Supported sources are `narratio.input.party`, `narratio.input.players`,
|
||||
`narratio.input.glossary`, and `narratio.input.spell_catalog`. Each map entry is
|
||||
required by its presence: omit a binding when the selected Notarius pipeline
|
||||
does not need it. A spell-catalog binding additionally requires an effective
|
||||
campaign or session `spell_catalog_file`.
|
||||
|
||||
Selectors accept Notarius's `slot`, `chunk.slot`, `lane.slot`,
|
||||
`lane.extract.slot`, `lane.merge.slot`, and `lane.normalize.slot` forms.
|
||||
Narratio trims whitespace around
|
||||
selectors and their dot-separated components, rejects empty components and
|
||||
`=`, rejects duplicate normalized selectors, and limits the map to 256 entries.
|
||||
It validates only selector structure and the prepared source vocabulary;
|
||||
Notarius owns target-slot declarations and media compatibility.
|
||||
|
||||
Before extraction, Narratio resolves every binding from the current prepared
|
||||
session manifest and streams it into a verified invocation-local snapshot whose
|
||||
absolute path is passed to Notarius. Missing, unsafe, empty,
|
||||
changed-during-copy, or checksum-inconsistent prepared evidence fails with
|
||||
guidance to force `prepare`. Bindings are sorted by normalized selector and are
|
||||
part of extraction fingerprint and resume identity. See the
|
||||
[Notarius integration contract](./integrations/notarius.md) for the subprocess
|
||||
boundary and the [complete example](../examples/pipeline.full.annotated.yml)
|
||||
for a copyable configuration.
|
||||
|
||||
### Notarius Output Entries
|
||||
|
||||
@@ -248,7 +310,7 @@ For each `pipeline.scriptorium.artifacts.<name>`:
|
||||
| Field | Type | Required | Rule |
|
||||
| --- | --- | --- | --- |
|
||||
| `enabled` | bool | No | `false` if omitted |
|
||||
| `depends_on[]` | list[string] | No | must reference configured artifact keys; no self-reference; enabled graph must be acyclic |
|
||||
| `depends_on[]` | list[string] | No | must reference configured artifact keys; no self-reference; configured graph must be acyclic |
|
||||
| `render_debug` | bool | No | per-artifact override |
|
||||
| `prompt_id` | string | Conditional | required when artifact is enabled |
|
||||
| `profile_id` | string | No | empty |
|
||||
@@ -259,34 +321,55 @@ For each `pipeline.scriptorium.artifacts.<name>`:
|
||||
|
||||
Narratio adds `session_id=narratio-session-<session_id>` to every Scriptorium request for sticky upstream LLM routing. If an artifact config sets `vars.session_id`, Narratio replaces that value before invoking Scriptorium. Use a different variable name if a prompt needs the raw Narratio session ID as content.
|
||||
|
||||
Without `--artifacts`, analyze executes enabled configured artifacts. With an
|
||||
explicit `--artifacts` list, the exact named configured artifacts are the
|
||||
one-invocation targets even if their `enabled` values are false. Analyze closes
|
||||
those targets over `depends_on`: a current prerequisite is reused, while a
|
||||
stale, missing, failed, or legacy prerequisite is rebuilt before its dependent.
|
||||
Unrelated artifacts are not executed. Named targets and any prerequisite that
|
||||
may require rebuilding must therefore have valid executable fields. This
|
||||
override affects analyze planning only; publish uses the list only to filter
|
||||
configured `narratio.artifact.<name>` output rules.
|
||||
|
||||
For each artifact input `pipeline.scriptorium.artifacts.<name>.inputs.<input_name>`:
|
||||
|
||||
| Field | Type | Required | Rule |
|
||||
| --- | --- | --- | --- |
|
||||
| `source` | string | Yes | built-in runtime source, prepared input source, `narratio.extraction.<name>`, `narratio.artifact.<name>`, or `narratio.previous_session.artifact.<name>` |
|
||||
| `artifact` | string | No | optional passthrough adapter field |
|
||||
| `path` | string | No | optional passthrough adapter field |
|
||||
| `required` | bool | No | optional input requirement |
|
||||
|
||||
`artifact` and `path` are obsolete and rejected by strict configuration
|
||||
loading. Use the canonical `source` identifier to select the input; Narratio
|
||||
does not provide adapter-specific input passthrough fields.
|
||||
|
||||
### Notifications
|
||||
|
||||
Narratio currently supports only `notification.mode: noop`, which is also the
|
||||
default when the section is omitted. The notify stage performs no delivery in
|
||||
this mode. Backend, recipient, timeout, and other provider settings are
|
||||
rejected by strict configuration loading until Narratio has a provider
|
||||
integration.
|
||||
|
||||
### Campaign
|
||||
|
||||
| Field | Type | Required | Notes |
|
||||
| --- | --- | --- | --- |
|
||||
| `campaign_id` | string | Yes | canonical campaign identity |
|
||||
| `campaign_id` | string | Yes | canonical opaque campaign identity |
|
||||
| `session_template_file` | string | No | used by `session init` when set |
|
||||
| `inputs.speakers_file` | string | Yes | stable input default |
|
||||
| `inputs.autocorrect_file` | string | Yes | stable input default |
|
||||
| `inputs.glossary_file` | string | Yes | stable input default |
|
||||
| `inputs.players_file` | string | Yes | stable input default |
|
||||
| `inputs.party_file` | string | Yes | stable input default |
|
||||
| `inputs.spell_catalog_file` | string | No | optional spell-catalog overlay default; required when a Notarius reference selects `narratio.input.spell_catalog` |
|
||||
|
||||
### Session
|
||||
|
||||
| Field | Type | Required in session file | Notes |
|
||||
| --- | --- | --- | --- |
|
||||
| `session_id` | string | Yes | must match CLI session target when provided |
|
||||
| `previous_session_id` | string | No | must not equal `session_id` |
|
||||
| `campaign` | string | No | filled from `campaign_id` during resolve if omitted |
|
||||
| `session_id` | string | Yes | opaque identity; must match CLI session target when provided |
|
||||
| `previous_session_id` | string | No | opaque identity; must not equal `session_id` |
|
||||
| `campaign` | string | No | opaque identity; filled from `campaign_id` during resolve if omitted |
|
||||
| `date` | string | No | metadata |
|
||||
| `title` | string | No | metadata |
|
||||
| `inputs.speakers_file` | string | No | overrides campaign stable input |
|
||||
@@ -294,6 +377,7 @@ For each artifact input `pipeline.scriptorium.artifacts.<name>.inputs.<input_nam
|
||||
| `inputs.glossary_file` | string | No | overrides campaign stable input |
|
||||
| `inputs.players_file` | string | No | overrides campaign stable input |
|
||||
| `inputs.party_file` | string | No | overrides campaign stable input |
|
||||
| `inputs.spell_catalog_file` | string | No | overrides the optional campaign spell catalog; empty or omitted inherits the campaign value |
|
||||
| `inputs.audio_dir` | string | Conditional | local audio mode |
|
||||
| `inputs.audio_files[]` | list[string] | Conditional | local audio mode |
|
||||
| `inputs.audio_s3.prefix` | string | Conditional | S3 audio mode |
|
||||
@@ -301,6 +385,20 @@ For each artifact input `pipeline.scriptorium.artifacts.<name>.inputs.<input_nam
|
||||
Audio rules:
|
||||
|
||||
- configure local mode (`audio_dir` or `audio_files`) or S3 mode (`audio_s3.prefix`), not both.
|
||||
- `audio_s3` requires `pipeline.storage.backend: s3` and a configured S3 bucket.
|
||||
|
||||
### Storage backend selection
|
||||
|
||||
`local` is the default and disables remote object-store operations. Configure
|
||||
`s3` explicitly before supplying `storage.s3`; a populated S3 block does not
|
||||
select a backend on its own. Unknown backend names and an S3 block paired with
|
||||
`local` are rejected during configuration validation.
|
||||
|
||||
### Previous-session expectation
|
||||
|
||||
`previous_session_id` is optional in a session file. When a command supplies
|
||||
`--previous-session-id`, however, the session file must contain the same value;
|
||||
an omitted or different value is rejected before the command performs work.
|
||||
|
||||
## Maintained Examples
|
||||
|
||||
|
||||
@@ -25,19 +25,34 @@ polished transcripts and generated artifacts. Start with the
|
||||
| Adapters or external tool contracts | [Adapter Internals](internal/adapters.md) and [Integration Contracts](integrations/README.md) | The internal guide owns adapter composition and mechanics; integration documents own external formats and protocols. |
|
||||
| Manifests, artifacts, workspace paths, or publish behavior | [Manifest Internals](internal/manifest.md), [Artifact Internals](internal/artifacts.md), [Workspace Internals](internal/workspace.md), [Publish Internals](internal/stage-publish.md), and [Operations](operations.md) | These separate implementation state and resolution from operator-visible layout and lifecycle. |
|
||||
| Maintained configuration or input examples | [Configuration](config.md) and [Examples](../examples/README.md) | The reference owns field meanings; the examples directory owns complete copyable files. |
|
||||
| Proposed or unimplemented behavior | [Roadmap](roadmap/) | Future work belongs only in roadmap documentation until implemented. |
|
||||
| Proposed or unimplemented behavior | `docs/roadmap/` | Future work belongs only in roadmap documentation until implemented. |
|
||||
|
||||
For an existing subsystem, also inspect its focused tests and package-level
|
||||
contracts before changing behavior.
|
||||
|
||||
## Validation
|
||||
|
||||
Use focused package tests while iterating. Run the repository-wide checks when a
|
||||
change affects shared contracts, application behavior, or maintained
|
||||
documentation examples:
|
||||
Use focused package tests while iterating. Every pull request and push runs the
|
||||
following repository-wide checks before it can be accepted:
|
||||
|
||||
```sh
|
||||
go test ./...
|
||||
go test -race ./...
|
||||
go vet ./...
|
||||
go build ./cmd/narratio
|
||||
go build ./...
|
||||
go test ./internal/doccheck
|
||||
go test ./internal/config -run '^TestExamplesLoadAndValidate$'
|
||||
```
|
||||
|
||||
The documentation check verifies local Markdown links and the dependency graph
|
||||
of the Woodpecker workflows. The configuration check loads every maintained
|
||||
pipeline and session example. Release automation repeats these checks and
|
||||
cross-compiles the CLI before it builds release assets; publishing depends on
|
||||
that validation path, so a failure cannot publish a release.
|
||||
|
||||
Woodpecker also runs `go test -race -shuffle=on -count=3 ./...` on its scheduled
|
||||
job to expose ordering and repeatability defects. Current runners cross-compile
|
||||
for macOS and Windows, but do not provide native macOS or Windows execution.
|
||||
Those cross-builds establish compilation only, not platform-equivalent runtime
|
||||
evidence. Add native checks only when official runner labels and successful
|
||||
native-run evidence are available.
|
||||
|
||||
@@ -15,7 +15,12 @@ runner composition is documented in
|
||||
- required transcript/glossary/output/work-dir paths;
|
||||
- optional report path (required when report mode is enabled);
|
||||
- generated config and stdout/stderr log paths;
|
||||
- optional module/model/base-url/config/output-schema/concurrency settings.
|
||||
- optional per-invocation module override.
|
||||
|
||||
The constructed runner owns static Audita settings: binary, timeout,
|
||||
credentials, default modules, model and endpoint settings, validation and output
|
||||
settings, report mode, and concurrency. The `polish` stage supplies only
|
||||
invocation-specific paths and may override modules for that invocation.
|
||||
|
||||
## Result Contract
|
||||
`PolishResult` returns:
|
||||
@@ -41,6 +46,9 @@ Run fails for:
|
||||
- invalid processed transcript JSON (`segments` array required);
|
||||
- invalid report JSON when reporting is enabled.
|
||||
|
||||
Processed transcript JSON is limited to 64 MiB and optional report JSON to 16
|
||||
MiB. Both must be regular files without symlinked path components.
|
||||
|
||||
Failure results still include output/log/config/exit metadata for diagnostics.
|
||||
|
||||
## Deterministic Behavior
|
||||
|
||||
@@ -7,12 +7,14 @@ lanes from the final trimmed Seriatim transcript. Narratio owns invocation,
|
||||
safe bundle discovery, lane selection, and its own artifact metadata. Notarius
|
||||
owns pipeline definitions, lane schemas, the receipt, and bundle formats.
|
||||
|
||||
Canonical Notarius references:
|
||||
Canonical Notarius v0.6.0 references:
|
||||
|
||||
- [Subprocess consumer contract](https://gitea.maximumdirect.net/eric/notarius/src/branch/main/docs/consumers/subprocess.md)
|
||||
- [D&D pipeline and lane contracts](https://gitea.maximumdirect.net/eric/notarius/src/branch/main/docs/consumers/dnd-pipeline.md)
|
||||
- [Run-result receipt](https://gitea.maximumdirect.net/eric/notarius/src/branch/main/docs/integrations/run-result.md)
|
||||
- [JSON output bundle](https://gitea.maximumdirect.net/eric/notarius/src/branch/main/docs/integrations/json-output.md)
|
||||
- [CLI reference](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/cli.md)
|
||||
- [Subprocess consumer contract](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/consumers/subprocess.md)
|
||||
- [D&D pipeline and lane contracts](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/consumers/dnd-pipeline.md)
|
||||
- [Run-result receipt](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/integrations/run-result.md)
|
||||
- [JSON output bundle](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/integrations/json-output.md)
|
||||
- [D&D spell-catalog overlay](https://gitea.maximumdirect.net/eric/notarius/src/tag/v0.6.0/docs/integrations/dnd-spell-catalog-overlays.md)
|
||||
|
||||
The [complete Narratio example](../../examples/pipeline.full.annotated.yml)
|
||||
records the exact current constraints for all ten D&D lanes. Treat the linked
|
||||
@@ -23,39 +25,77 @@ duplicate the complete schemas.
|
||||
|
||||
When `pipeline.notarius.enabled` is true, Narratio resolves the executable,
|
||||
configuration path, input path, output directory, and working directory to
|
||||
absolute paths and invokes:
|
||||
absolute paths. Narratio requires the Notarius v0.6.0 CLI contract when
|
||||
references are configured and invokes each binding as a separate argument
|
||||
before `--json`:
|
||||
|
||||
```text
|
||||
notarius run <pipeline_id> --config <config_path> --input <trimmed_json> --output-dir <staging_dir> --json
|
||||
notarius run <pipeline_id> --config <config_path> --input <trimmed_json> --output-dir <staging_dir> [--reference <selector>=<verified_snapshot_path>]... --json
|
||||
```
|
||||
|
||||
Reference paths are absolute invocation-local snapshots streamed from the
|
||||
manifest-verified canonical files prepared inside the current Narratio session
|
||||
workspace. Narratio verifies snapshot checksum and size before and after the
|
||||
subprocess, and passes only configured bindings, ordered lexically by normalized
|
||||
selector, as direct argument-vector entries without shell interpretation. A CLI
|
||||
binding takes precedence over a matching external path in Notarius
|
||||
configuration. Narratio never emits `--without-reference`.
|
||||
|
||||
The maintained D&D boundary binds only the four campaign-owned external slots:
|
||||
|
||||
```text
|
||||
notarius run dnd-session \
|
||||
--config <absolute config path> \
|
||||
--input <absolute trimmed transcript path> \
|
||||
--output-dir <absolute staging directory> \
|
||||
--reference glossary=<absolute verified glossary snapshot> \
|
||||
--reference party=<absolute verified party snapshot> \
|
||||
--reference players=<absolute verified players snapshot> \
|
||||
--reference spell_catalog=<absolute verified spell catalog snapshot> \
|
||||
--json
|
||||
```
|
||||
|
||||
The spell-catalog binding is omitted when the campaign does not maintain that
|
||||
optional overlay. Registry, scene-description, combat-turn, and occurrence
|
||||
handoffs generated during the same Notarius run remain in Notarius pipeline
|
||||
composition and must not be emitted as CLI references. The linked CLI and D&D
|
||||
consumer documents own selector targeting, declared slots, media compatibility,
|
||||
and generated-handoff collision rules.
|
||||
|
||||
Standard output is reserved for the JSON receipt. Standard error is captured
|
||||
separately as diagnostic output. Narratio applies the configured timeout and
|
||||
does not interpret stdout as a receipt unless the subprocess exits successfully.
|
||||
It does not pass a Narratio session ID or run `notarius config validate`
|
||||
automatically; the configured working directory and inherited environment
|
||||
apply to the subprocess.
|
||||
automatically; the configured working directory and Narratio's minimal child
|
||||
environment apply to the subprocess.
|
||||
|
||||
## Accepted Result
|
||||
|
||||
Narratio currently accepts receipt schema `notarius.run-result.v1`. The receipt
|
||||
Narratio's supported invocation baseline is Notarius v0.6.0. The accepted
|
||||
receipt remains `notarius.run-result.v2`; reference flags do not change the
|
||||
receipt or ten-lane output contract. The receipt
|
||||
must identify the configured pipeline, and its `index_file` must be exactly
|
||||
`index.json` beneath the reported bundle root. The production index must name
|
||||
the management files exactly as `manifest.json`, `rejected.json`, and
|
||||
`warnings.json`. All receipt, index, and lane paths must stay inside that
|
||||
bundle; symlinks and non-regular lane payloads are rejected.
|
||||
the management files exactly as `manifest.json`, `rejected.json`,
|
||||
`warnings.json`, and `diagnostics.json`. All receipt, index, and lane paths must
|
||||
stay inside that bundle; symlinks and non-regular lane payloads are rejected.
|
||||
|
||||
Supported receipt and index shapes tolerate unknown fields for forward
|
||||
compatibility, while required identity, validation, count, manifest,
|
||||
rejection, warning, and lane-list fields remain mandatory. Narratio applies
|
||||
bounded reads to the receipt, index, rejection, and warning documents. Optional
|
||||
chunk-map and evidence-context descriptors must carry their complete generic
|
||||
contract metadata when present.
|
||||
rejection, warning, diagnostic, and lane-list fields remain mandatory.
|
||||
Narratio applies bounded reads to the receipt, index, rejection, warning, and
|
||||
diagnostic documents. Warning and diagnostic envelopes, group counts,
|
||||
occurrence counts, truncation state, framework-owned origins, and
|
||||
receipt-to-bundle counts must be internally consistent. Optional chunk-map and
|
||||
evidence-context descriptors must carry their complete generic contract
|
||||
metadata when present.
|
||||
|
||||
For every entry in `pipeline.notarius.outputs`, Narratio requires exactly one
|
||||
index descriptor with the configured lane ID, media type, schema ID, schema
|
||||
version, and, when configured, module key. Missing, duplicate, rejected, or
|
||||
incompatible required lanes fail extraction even if Notarius exited zero.
|
||||
incompatible required lanes fail extraction even if Notarius exited zero. A
|
||||
configured lane whose v2 validation summary is `rejected` or `incomplete` also
|
||||
fails extraction.
|
||||
Unconfigured lanes may remain in the preserved bundle but do not become
|
||||
selectable Narratio sources.
|
||||
|
||||
@@ -72,10 +112,13 @@ only explicitly named lane sources; `--artifacts` never selects Notarius lanes.
|
||||
staged bundle is promoted to durable storage.
|
||||
- Contract and external provenance metadata are preserved on lane artifact
|
||||
records and through explicit publication.
|
||||
- Undeclared selectors, incompatible reference files, and external/generated
|
||||
reference collisions are Notarius errors and fail extraction normally.
|
||||
|
||||
Rejection and warning summaries retain structured stage, scope, lane, and
|
||||
reason-code fields for diagnostics without exposing free-form external messages
|
||||
or reading lane payload bodies.
|
||||
Rejection, validation, warning, and diagnostic summaries retain bounded stable
|
||||
identity, category, origin, reason-code, status, and occurrence fields without
|
||||
copying free-form external messages into Narratio manifest metadata or reading
|
||||
lane payload bodies.
|
||||
|
||||
Configuration fields and defaults are in [Configuration](../config.md).
|
||||
Operator paths, rerun procedures, and bundle retention are in
|
||||
|
||||
@@ -46,6 +46,9 @@ Run behavior:
|
||||
- `run` exit code `2` is mapped to `ValidationFailed=true`;
|
||||
- successful subprocess still fails if output file is missing or empty.
|
||||
|
||||
Each artifact result is limited to 64 MiB and must be a regular file without
|
||||
symlinked path components.
|
||||
|
||||
Render behavior:
|
||||
- subprocess errors propagate;
|
||||
- output file must exist and be non-empty.
|
||||
|
||||
@@ -41,6 +41,8 @@ Invocation fails on:
|
||||
- empty render output files.
|
||||
|
||||
When report paths are provided/enabled, report files must parse as JSON.
|
||||
Each Seriatim JSON or rendered-text result is limited to 64 MiB and must be a
|
||||
regular file without symlinked path components.
|
||||
|
||||
## Deterministic Behavior
|
||||
- argument ordering is deterministic per command construction.
|
||||
|
||||
@@ -17,6 +17,11 @@ Narratio sends an HTTP `POST` to the configured transcription URL using
|
||||
The server must return a `2xx` response whose body is valid JSON. Narratio does
|
||||
not currently require a more specific response schema at this boundary.
|
||||
|
||||
The transcription URL must be an absolute `http` or `https` URL. The audio body
|
||||
is streamed through a fresh multipart writer for every attempt, so its memory
|
||||
use is bounded by the transport buffer rather than by the complete audio file.
|
||||
WhisperX response acquisition is capped at 10 MiB.
|
||||
|
||||
## Request And Result Contract
|
||||
|
||||
Each adapter request identifies a speaker, a readable audio file, and the
|
||||
@@ -40,7 +45,7 @@ transcript output.
|
||||
|
||||
## Validation And Failure Semantics
|
||||
|
||||
Client construction rejects a missing or invalid absolute transcription URL,
|
||||
Client construction rejects a missing or non-HTTP(S) absolute transcription URL,
|
||||
a missing language, a non-positive timeout, negative retries, or a negative
|
||||
retry delay. A request fails before transmission when its audio or output path
|
||||
is missing.
|
||||
|
||||
@@ -35,18 +35,33 @@ Adapters do not own:
|
||||
|
||||
## Default Wiring
|
||||
|
||||
`internal/app/runner.go` initializes default adapters when not injected:
|
||||
`internal/app/runner.go` initializes default adapters when not injected and
|
||||
only when the selected execution plan needs them:
|
||||
|
||||
- WhisperX HTTP client from pipeline config.
|
||||
- Seriatim subprocess runner.
|
||||
- Audita subprocess runner.
|
||||
- Scriptorium subprocess runner.
|
||||
- Notarius subprocess runner when extraction is enabled.
|
||||
- Noop notifier (`notify.NoopSender`).
|
||||
- WhisperX HTTP client for `transcribe`.
|
||||
- Seriatim subprocess runner for `merge`, `normalize`, `trim`, or `render`.
|
||||
- Audita subprocess runner for `polish`.
|
||||
- Scriptorium subprocess runner for `trim` or `analyze`.
|
||||
- Notarius subprocess runner for `extract` when extraction is enabled.
|
||||
- Noop notifier (`notify.NoopSender`) for `notify`.
|
||||
- Object store only when required by selected stages/config.
|
||||
|
||||
Remote publish locks are loaded only for a selected, enabled publish that
|
||||
uploads a run. Shared session lifecycle setup still applies to every selected
|
||||
range, but an unselected integration is neither initialized nor validated by
|
||||
runner composition. Each selected stage retains its own fail-fast configuration
|
||||
and input validation.
|
||||
|
||||
`session plan` is outside production adapter composition. It performs
|
||||
resume validation and models selected transitions against cloned manifest
|
||||
state without constructing or invoking stage-execution adapters. The shared
|
||||
command configuration loader may still use object storage to retrieve a missing
|
||||
remote session file before planning begins.
|
||||
|
||||
Notarius is composed only when extraction is enabled; the extract stage owns
|
||||
receipt, bundle, and configured-lane policy rather than the adapter.
|
||||
prepared reference resolution, receipt, bundle, and configured-lane policy.
|
||||
The adapter validates the ordered selector/absolute-path pairs and is the sole
|
||||
owner of serializing them as repeated `--reference` arguments before `--json`.
|
||||
|
||||
Object-store construction goes through `newCommandObjectStore`, which loads
|
||||
configured filesystem secrets before adapter initialization.
|
||||
@@ -56,6 +71,18 @@ configured filesystem secrets before adapter initialization.
|
||||
- Constructor errors fail stage execution setup early.
|
||||
- Runtime adapter errors propagate to stage code and then manifest failure handling.
|
||||
- Subprocess adapters persist stage logs/generated configs through stage-managed paths.
|
||||
- Shared subprocess execution starts an owned process group on Linux/macOS or a
|
||||
kill-on-close job object on Windows. Every terminal path disposes of that
|
||||
owned tree before returning. After a natural leader exit, Unix checks for
|
||||
remaining group members and uses bounded graceful then forceful termination;
|
||||
Windows closes the job so kill-on-close applies. Cancellation, deadlines, and
|
||||
diagnostic limits use the same terminal disposal path without losing their
|
||||
original result classification. Child environments contain only the execution
|
||||
baseline and adapter-specified values; configured credentials are explicit
|
||||
sensitive values. Stdout and stderr are redacted while streaming into separate
|
||||
8 MiB diagnostic captures; a bounded wait closes a stream retained by a
|
||||
departed leader's descendant. Unsupported platforms reject owned command
|
||||
execution.
|
||||
|
||||
## Implementation And Tests
|
||||
|
||||
|
||||
@@ -30,23 +30,51 @@ and content validator. The focused stage documents own their input/output flow;
|
||||
- extraction source ID format: `narratio.extraction.<output_key>`
|
||||
- previous-session source ID format: `narratio.previous_session.artifact.<artifact_key>`
|
||||
|
||||
All formats are validated by strict source-policy rules. Extraction sources are
|
||||
registered only from `pipeline.notarius.outputs`; the Notarius index has no
|
||||
selectable source ID.
|
||||
All formats are validated by strict source-policy rules. Configured artifact and
|
||||
extraction keys use `^[a-z][a-z0-9_]*$`; source parsers never normalize an
|
||||
unrecognized token into a valid source. Extraction sources are registered only
|
||||
from `pipeline.notarius.outputs`; the Notarius index has no selectable source
|
||||
ID.
|
||||
|
||||
Prepared stable source IDs are `narratio.input.players`,
|
||||
`narratio.input.party`, `narratio.input.glossary`, and
|
||||
`narratio.input.spell_catalog`. Artifact policy owns their canonical manifest
|
||||
kind and prepared filename vocabulary.
|
||||
|
||||
## Runtime Catalog
|
||||
|
||||
`ArtifactCatalog` tracks:
|
||||
|
||||
- `planned`: source registered for run context;
|
||||
- `executable`: selected and enabled for analyze execution;
|
||||
- `available`: local file exists and validates;
|
||||
- `executable`: included in the effective analyze artifact set;
|
||||
- `available`: the source's canonical evidence owner validates its current
|
||||
manifest record and durable bytes;
|
||||
- `provenance`: availability source.
|
||||
|
||||
Configured definitions are always registered. Without an explicit selection,
|
||||
the effective analyze set contains enabled definitions. With `--artifacts`, the
|
||||
exact named configured definitions become the effective set for that invocation,
|
||||
regardless of their `enabled` value. The effective-set resolver itself does not
|
||||
expand dependencies; the analyze work planner closes those targets over their
|
||||
configured prerequisite graph. Availability is separate from executability.
|
||||
Configured outputs, including non-executable prerequisites, become available
|
||||
only when the versioned analyze state identifies a current result whose source,
|
||||
contract, canonical configured path, size, and checksum match a confined
|
||||
no-follow regular file. An incidental canonical file and a legacy aggregate
|
||||
analyze output are unavailable.
|
||||
Extraction entries are registered from configuration and become available only
|
||||
after compatible extraction evidence is hydrated.
|
||||
|
||||
During an analyze invocation, a newly validated and atomically materialized
|
||||
configured output is marked available with its producer run ID, contract,
|
||||
checksum, and size. Later scheduled dependents therefore observe the same
|
||||
semantic identity whether their prerequisite was reused from current manifest
|
||||
evidence or produced earlier in the invocation.
|
||||
|
||||
Current provenance values:
|
||||
|
||||
- `generated.current_analyze_run`
|
||||
- `filesystem.disabled_artifact_output`
|
||||
- `manifest.current_analyze_artifact`
|
||||
- `manifest.inputs.previous_cache`
|
||||
- `current_session.previous_cache`
|
||||
|
||||
@@ -59,15 +87,37 @@ Built-ins:
|
||||
|
||||
Configured sources (`narratio.artifact.*`):
|
||||
|
||||
- resolve only through runtime catalog availability.
|
||||
- resolve only through runtime catalog availability;
|
||||
- use the shared typed analyze-evidence inspection in
|
||||
`analyze_evidence.go` for prior current-session results;
|
||||
- require the supported analyze-state and fingerprint versions, a `current`
|
||||
record for the exact configured key and source ID, a complete contract, the
|
||||
configured canonical relative path, positive stored size, and stored
|
||||
checksum matching bytes read from a confined no-follow regular file; and
|
||||
- treat non-current statuses, legacy or malformed records, removed keys,
|
||||
unsafe or missing files, and size/checksum mismatches as unavailable without
|
||||
rewriting manifest state. Catalog construction iterates current
|
||||
configuration, so removed or renamed records are not advertised.
|
||||
|
||||
Prepared stable sources (`narratio.input.*`):
|
||||
|
||||
- resolve only from the current manifest's exact prepared-input record;
|
||||
- require the policy-owned canonical path below the session root, a confined
|
||||
non-symlink regular file, a non-empty payload, and a matching SHA-256
|
||||
checksum; and
|
||||
- return an immutable source/path/checksum/size identity shared by extract and
|
||||
analyze rather than falling back to campaign/session source paths.
|
||||
|
||||
Extraction sources (`narratio.extraction.*`):
|
||||
|
||||
- use the shared registration and manifest hydration path in
|
||||
`extraction_catalog.go`;
|
||||
- use the shared typed bundle evidence inspection in `extraction_evidence.go`;
|
||||
- require a current successful extract record with the exact configured source,
|
||||
compatible contract and Notarius provenance, a confined regular durable
|
||||
payload, and matching checksum; and
|
||||
payload, matching checksum, and the current resolved trimmed-transcript
|
||||
identity;
|
||||
- remain unavailable unless catalog hydration receives valid evidence. Resume
|
||||
treats absent or obsolete evidence as a rerun decision and unsafe evidence as
|
||||
an error; and
|
||||
- are never inferred by scanning the Notarius bundle directory.
|
||||
|
||||
Previous-session sources (`narratio.previous_session.artifact.*`):
|
||||
@@ -76,6 +126,10 @@ Previous-session sources (`narratio.previous_session.artifact.*`):
|
||||
- prefer manifest-backed previous-input paths;
|
||||
- fallback to existing previous-cache filesystem paths.
|
||||
|
||||
Source absence is evaluated by the consuming artifact input. An optional input
|
||||
is omitted from that invocation; a required input fails resolution. This is
|
||||
separate from a stage's lifecycle outcome.
|
||||
|
||||
Validation by content type:
|
||||
|
||||
- transcript JSON built-ins: JSON with top-level `segments` array;
|
||||
@@ -87,7 +141,7 @@ Validation by content type:
|
||||
|
||||
`CollectPreviousArtifactRequirements`:
|
||||
|
||||
- scans enabled configured artifacts only;
|
||||
- scans the effective configured artifact set;
|
||||
- extracts only canonical previous-session sources;
|
||||
- deduplicates by artifact key;
|
||||
- merges required and optional references (required wins);
|
||||
@@ -98,12 +152,31 @@ Validation by content type:
|
||||
Artifacts package owns shared remote current-state loading mechanics used by
|
||||
restore, status and validation checks, and previous-cache planning.
|
||||
|
||||
For a new-protocol current state, the pointer-selected immutable commit is the
|
||||
complete restore authority. Callers receive its declared object identities and
|
||||
must not supplement them by listing mutable session prefixes. The legacy reader
|
||||
is intentionally separate and remains migration-only support.
|
||||
|
||||
The reader opens each small control object directly and enforces owner-specific
|
||||
limits before decoding: 64 KiB for the mutable commit pointer, 4 MiB for the
|
||||
immutable commit manifest, and 8 MiB for the selected session manifest. Legacy
|
||||
compatibility applies a 4 KiB limit to `current/run_id.txt` and the same 8 MiB
|
||||
manifest limit to `current/manifest.json`. These are exposed as
|
||||
`MaxCurrentCommitPointerBytes`, `MaxRemoteCommitManifestBytes`,
|
||||
`MaxRemoteSessionManifestBytes`, `MaxLegacyCurrentRunPointerBytes`, and
|
||||
`MaxLegacyCurrentManifestBytes`.
|
||||
|
||||
Each read uses the generation and size metadata returned with its opened body.
|
||||
Actual bytes remain subject to a limit-plus-one read even if size metadata is
|
||||
absent or inaccurate. Immutable selections then retain their declared-size,
|
||||
checksum, generation, and identity checks. No current-state control object is
|
||||
downloaded through a temporary file.
|
||||
|
||||
Core helpers:
|
||||
|
||||
- `LoadCurrentRunPointer`
|
||||
- `LoadCurrentManifest`
|
||||
- `LoadCurrentState`
|
||||
- `ValidateCurrentStateIdentity`
|
||||
- `RemoteCommitManifest` and `CurrentCommitPointer`
|
||||
|
||||
Typed missing-state errors:
|
||||
|
||||
@@ -131,6 +204,18 @@ Caller policy is intentionally outside artifacts helpers:
|
||||
- spool/cache paths;
|
||||
- S3 session/run/current-state key layout.
|
||||
|
||||
New publication creates run-scoped immutable objects, including
|
||||
`runs/{run_id}/commit.json` and `runs/{run_id}/session-manifest.json`. The sole
|
||||
mutable selector is `current/commit-pointer.json`; readers verify its selected
|
||||
commit and declared object generations/checksums. Legacy current-pair loading
|
||||
is confined to `current_state_legacy.go` for migration only.
|
||||
|
||||
Campaign, session, and Narratio run IDs are validated as portable opaque
|
||||
segments at configuration and artifact boundaries before they can be used in a
|
||||
workspace or S3 namespace. Previous-artifact destinations remain typed,
|
||||
multi-segment relative paths and are confined beneath `previous/artifacts`; they
|
||||
are not treated as opaque identifiers.
|
||||
|
||||
See [Workspace Internals](workspace.md) for how callers consume local helpers
|
||||
and [Operations](../operations.md#local-state-layout) for the authoritative
|
||||
physical layout.
|
||||
@@ -148,8 +233,13 @@ physical layout.
|
||||
|
||||
- Registry and resolution: `internal/artifacts/artifact_resolver.go`,
|
||||
`internal/artifacts/catalog.go`, `internal/artifacts/transcripts.go`,
|
||||
`internal/artifacts/extraction_catalog.go`
|
||||
- Current state: `internal/artifacts/current_state.go`
|
||||
`internal/artifacts/extraction_catalog.go`,
|
||||
`internal/artifacts/extraction_evidence.go`,
|
||||
`internal/artifacts/extraction_input.go`,
|
||||
`internal/artifacts/prepared_input.go`
|
||||
- Current state: `internal/artifacts/current_state.go`,
|
||||
`internal/artifacts/current_state_commit.go`,
|
||||
`internal/artifacts/current_state_legacy.go`
|
||||
- Paths and keys: `internal/artifacts/paths.go`,
|
||||
`internal/artifacts/s3_keys.go`
|
||||
- Previous requirements: `internal/artifacts/previous_requirements.go`
|
||||
|
||||
@@ -8,8 +8,8 @@ reporting flow in `internal/app`. User invocation belongs in
|
||||
physical restore scope belong in
|
||||
[Operations](../operations.md#restore-workflow).
|
||||
|
||||
Restore is split into explicit phases so remote authority, local conflict
|
||||
policy, and filesystem mutation can be tested independently.
|
||||
Restore separates remote authority, local conflict policy, and filesystem
|
||||
mutation so each remains testable independently.
|
||||
|
||||
## Discovery Contract
|
||||
|
||||
@@ -18,6 +18,7 @@ Discovery delegates current-state pointer and manifest loading to
|
||||
|
||||
- campaign must match;
|
||||
- session ID must match.
|
||||
- run ID must match the pointer-selected committed run.
|
||||
|
||||
Restore treats any missing or invalid remote current state as a command error.
|
||||
|
||||
@@ -31,13 +32,20 @@ Restore planner action kinds:
|
||||
|
||||
Planner behavior:
|
||||
|
||||
- remote list scope is the resolved session prefix;
|
||||
- a new-protocol restore uses only the selected commit's declared artifact set;
|
||||
each action carries that artifact's immutable key, checksum, size, and
|
||||
generation. Coherent legacy state remains on the isolated compatibility path;
|
||||
- remote-to-local mapping is traversal-safe;
|
||||
- actions are sorted by local relative path and then remote key;
|
||||
- force converts differing local targets from conflicts to downloads.
|
||||
- force converts differing eligible regular files from conflicts to downloads;
|
||||
directories and other non-regular targets remain conflicts.
|
||||
|
||||
Previous-cache files are planned separately through `previouscache.BuildPlan`
|
||||
when configured previous-session requirements exist.
|
||||
For a non-dry-run restore, planning/classification happens only after acquiring
|
||||
the session lock. Runner manifest/reuse checks acquire that same lock first.
|
||||
|
||||
Previous-cache readiness is resolved through `previouscache.Resolve` for restore,
|
||||
prepare, status, and validation. A committed source is selected only by its
|
||||
exact source identity; legacy fallback remains isolated and rejects ambiguity.
|
||||
|
||||
## Execution Contract
|
||||
|
||||
@@ -47,17 +55,32 @@ Execution order and safety:
|
||||
- `manifest.json` installs last;
|
||||
- downloads use sibling temp files plus atomic rename;
|
||||
- manifest replacement is validated before rename;
|
||||
- each committed object is verified against its declared checksum, size, and
|
||||
generation before installation;
|
||||
- a committed manifest already verified during discovery is retained for the
|
||||
matching restore action and revalidated before installation, avoiding a
|
||||
second body transfer;
|
||||
- failed installs do not roll back files already written in the same execution.
|
||||
- a durable `.restore-incomplete.json` marker is written before installation.
|
||||
It blocks runners until a restore retry completes all verified installs and
|
||||
the local manifest replacement, at which point it is removed.
|
||||
- restored manifest local references are rebased beneath the selected local
|
||||
session root. Unsafe relative references and producer-machine absolute paths
|
||||
outside the manifest's producer session root are rejected; producer-local
|
||||
spool/cache and cleanup locations are not restored as authority.
|
||||
|
||||
Audio restore path:
|
||||
|
||||
- uses `audio.MaterializeS3Audio`;
|
||||
- integrates spool and S3 audio cache paths;
|
||||
- supports cache-hit reuse without object redownload.
|
||||
- reuses cached audio only when its no-follow regular file, content digest, and
|
||||
identity sidecar all match the selected remote object version; otherwise it
|
||||
refreshes through the durable download path.
|
||||
|
||||
## Reporting Contract
|
||||
|
||||
- dry-run mode prints a summary and performs no local writes;
|
||||
- dry-run mode prints a summary, performs no durable session writes, and may
|
||||
read remote current-state or object-identity data to produce that summary;
|
||||
- execution mode persists the canonical restore report described in
|
||||
[Operations](../operations.md#restore-workflow);
|
||||
- report includes plan counts, per-action status, and execution failures.
|
||||
@@ -65,7 +88,13 @@ Audio restore path:
|
||||
## Invariants
|
||||
|
||||
- restore uses committed remote current state as authority;
|
||||
- `current/run_id.txt` is the remote publish commit marker;
|
||||
- one restore or status inspection observes the single pointer-selected commit
|
||||
loaded at discovery; later pointer changes cannot add objects or substitute a
|
||||
different run into its plan;
|
||||
- a verified `current/commit-pointer.json` and its selected immutable commit
|
||||
establish new-protocol remote commitment; coherent legacy
|
||||
`current/run_id.txt` plus `current/manifest.json` remains read-only migration
|
||||
support;
|
||||
- restore does not execute pipeline stages.
|
||||
|
||||
## Implementation And Tests
|
||||
|
||||
71
docs/internal/fileops.md
Normal file
71
docs/internal/fileops.md
Normal file
@@ -0,0 +1,71 @@
|
||||
# Internal: File Operations
|
||||
|
||||
`internal/fileops` owns the narrow mechanics for durable replacement of one
|
||||
byte file. Callers keep ownership of serialization, validation, cancellation,
|
||||
and destination-directory policy.
|
||||
|
||||
## Destination Confinement
|
||||
|
||||
Before it creates, replaces, or installs a destination file, `fileops` opens
|
||||
each ancestor from the filesystem root and rejects symbolic links or components
|
||||
that change during traversal. The resulting parent-directory handle is retained
|
||||
for sibling temporary-file creation and rename, so a later pathname swap cannot
|
||||
redirect the replacement. Existing destination symlinks are replaced as leaf
|
||||
entries; their targets are never followed.
|
||||
|
||||
Remote object acquisition uses a writer supplied by the storage owner. The
|
||||
writer receives a `fileops`-owned, already-open sibling temporary file rather
|
||||
than a mutable destination path. Callers still own remote object selection,
|
||||
validation, conflict handling, and final mode.
|
||||
|
||||
Directory promotion keeps the verified destination parent open while it creates
|
||||
the temporary tree, copies regular source entries, and performs the platform
|
||||
no-replace rename. Platforms without a verified handle-relative atomic
|
||||
no-replace primitive reject promotion before writing a temporary tree.
|
||||
|
||||
## Cleanup Contract
|
||||
|
||||
`RemoveAllUnderRoot` accepts an explicit root and a proper descendant. It opens
|
||||
the root and each target ancestor without following symlinks, then removes the
|
||||
tree through those directory handles. It rejects root deletion and any symlink
|
||||
encountered in the target path or tree; repeated removal of a missing target is
|
||||
successful. Command and post-publish policy remains owned by `internal/app`.
|
||||
|
||||
## Confined Reads
|
||||
|
||||
`ReadRegularFileUnderRoot` is the no-follow, bounded read primitive for a
|
||||
caller-selected root and relative file path; `ReadRegularFile` is its
|
||||
path-based convenience wrapper. They verify every ancestor through directory
|
||||
handles and admit only a stable regular-file handle. Callers enforce their own
|
||||
byte limits and access policy. Credential mode policy and environment
|
||||
precedence remain owned by `internal/app`.
|
||||
|
||||
## Replacement Contract
|
||||
|
||||
`ReplaceFileAtomic` requires an existing destination directory. It creates a
|
||||
sibling temporary file, writes the complete byte sequence, applies the
|
||||
caller-supplied mode, syncs and closes the file, runs an optional pre-rename
|
||||
check, replaces the destination with a rename, then syncs the containing
|
||||
directory.
|
||||
|
||||
The pre-rename check is the last point at which a caller can cancel without
|
||||
installing a new destination. A failure before the rename leaves the old
|
||||
destination unchanged and removes the temporary file; any cleanup failure is
|
||||
returned alongside the primary failure. A failure after the rename may leave
|
||||
the new file visible, but it is not reported as crash-durable.
|
||||
|
||||
Replacement follows the operating system's same-filesystem rename semantics.
|
||||
If a platform cannot replace an existing destination, the operation returns an
|
||||
error and never removes the old file as an emulation step.
|
||||
|
||||
## Directory-Sync Support
|
||||
|
||||
Linux and macOS attempt to sync the destination directory. Windows opens the
|
||||
directory with backup semantics and flushes its buffers. If either operation
|
||||
is unavailable for the platform, directory handle, or filesystem,
|
||||
`ErrDirectorySyncUnsupported` is returned. Narratio does not treat that result
|
||||
as successful crash-durable replacement.
|
||||
|
||||
`WriteFileAtomic`, copy helpers, and downloaded temporary-file installation
|
||||
retain their compatibility behavior of creating the destination parent with
|
||||
the repository's workspace permissions before using this contract.
|
||||
@@ -16,6 +16,14 @@ Explain the session-progress and invocation-audit models implemented by
|
||||
- `inputs` records
|
||||
- durable `artifacts` records
|
||||
- per-stage `stages` map
|
||||
- an optional `post_publish_cleanup` obligation, which binds a committed run,
|
||||
remote commit identity, and each exact root-confined local target to its
|
||||
completion evidence
|
||||
|
||||
Session, campaign, and run identities in local and downloaded manifests must be
|
||||
portable opaque segments. Unsafe legacy identities are rejected with migration
|
||||
guidance rather than being normalized into a different workspace or remote
|
||||
namespace.
|
||||
|
||||
The model admits these stage states:
|
||||
|
||||
@@ -27,6 +35,79 @@ The model admits these stage states:
|
||||
- `stale`
|
||||
- `interrupted`
|
||||
|
||||
### Analyze-owned artifact state
|
||||
|
||||
The `analyze` stage record may carry `analyze_state_version: 1` and an
|
||||
`analyze_artifacts` map keyed by normalized configured artifact key. The
|
||||
version is the authority marker: version 1 with no entries is a valid evaluated
|
||||
empty set, while an absent version is legacy aggregate-only state and provides
|
||||
no current configured-artifact evidence.
|
||||
|
||||
Each analyze artifact record has one disposition:
|
||||
|
||||
- `current`: the configured artifact is available and carries a versioned
|
||||
fingerprint plus a complete output record and separate output size;
|
||||
- `stale`: the recorded semantic identity is no longer current;
|
||||
- `missing`: no validated current result exists;
|
||||
- `failed`: the attempted work failed and carries a bounded diagnostic; or
|
||||
- `unselected`: the artifact was intentionally outside the evaluated set.
|
||||
|
||||
Records bind their normalized key and dependencies, fingerprint contract when
|
||||
evaluated, canonical session-relative output identity when current, producing
|
||||
Narratio run, update time, and bounded non-secret Scriptorium provenance and
|
||||
diagnostic paths. A current output includes its configured source ID, contract,
|
||||
checksum, and positive byte size. Non-current records cannot carry an output,
|
||||
so an older file is not advertised through stale, missing, failed, or
|
||||
unselected state.
|
||||
|
||||
The session-stage collection is the reconciled authority across invocations.
|
||||
The corresponding collection on an invocation's `analyze` stage record is an
|
||||
audit of only the artifacts evaluated or attempted by that run. These records
|
||||
remain analyze-owned data inside the fixed stage; they are not dynamic stages
|
||||
or generic subtasks.
|
||||
|
||||
The stage result contract has one analyze-specific projection boundary. On
|
||||
success, the runner validates and deep-copies the complete reconciled session
|
||||
collection and the invocation subset. Aggregate session outputs are rebuilt in
|
||||
configured-key order from current session records only; invocation outputs are
|
||||
limited to current records produced by that invocation's run ID. Ordinary
|
||||
stage outputs cannot accompany this projection, so there is one source of
|
||||
artifact authority.
|
||||
|
||||
Successful incremental execution replaces only evaluated artifact records and
|
||||
preserves valid unrelated current records. Rebuilt outputs are compared by
|
||||
bytes and contract: an unchanged identity permits an unselected dependent with
|
||||
the same recomputed fingerprint to remain current, while a changed identity
|
||||
removes output authority from every unselected transitive dependent by marking
|
||||
it stale. A partial analyze invocation can therefore succeed while unrelated
|
||||
configured records remain stale. Existing canonical files never create current
|
||||
records without validated execution and projection.
|
||||
|
||||
Aggregate analyze status is deliberately coarser than this collection. Resume
|
||||
validation may skip a succeeded aggregate record when the selected artifact
|
||||
closure is current even if unrelated records are stale. Conversely, a stale
|
||||
aggregate record may cross the ordinary runner boundary and perform zero
|
||||
Scriptorium calls when reconciliation proves every selected artifact current;
|
||||
the successful projection then restores the aggregate status.
|
||||
|
||||
Analyze may return a projection together with an error. That restricted result
|
||||
cannot carry ordinary outputs, skip state, aggregate logs, generated configs,
|
||||
or metadata. The runner persists only the validated per-artifact collections,
|
||||
then marks the aggregate analyze and run state failed and invalidates delivery
|
||||
dependents conservatively. Unrelated current records survive because the
|
||||
session projection is complete. A malformed projection is not applied, and a
|
||||
failed session projection save restores the prior per-artifact authority before
|
||||
terminal failure persistence.
|
||||
|
||||
The incremental executor constructs this restricted projection at each
|
||||
scheduled artifact boundary. The active record is failed without output,
|
||||
current transitive dependents are stale, unrelated current records survive, and
|
||||
only earlier validated and materialized completions remain current in the
|
||||
invocation subset. Session failure state is persisted before invocation failure
|
||||
state. If either terminal save fails, its persistence error is joined with the
|
||||
original adapter, validation, or filesystem cause; a failed projection save
|
||||
does not turn incidental canonical bytes into manifest authority.
|
||||
|
||||
## Run Manifest
|
||||
|
||||
`manifest.RunManifest` is created for each invocation and records:
|
||||
@@ -37,15 +118,40 @@ The model admits these stage states:
|
||||
- per-stage status
|
||||
- overall run status (`running`, `succeeded`, `failed`)
|
||||
|
||||
## Remote Commit Manifest
|
||||
|
||||
`artifacts.RemoteCommitManifest` is a separate, versioned remote snapshot
|
||||
contract. It is not a serialized session manifest and contains no local
|
||||
post-publication assertion such as `current_pointer_written`. A remote commit
|
||||
identifies one campaign, session, and run and declares its immutable artifact
|
||||
set. Each artifact has a typed source, immutable destination key, SHA-256
|
||||
checksum, size, and storage generation.
|
||||
|
||||
`current/commit-pointer.json` is the sole mutable selector for the new
|
||||
contract. It identifies exactly one run-scoped `runs/{run_id}/commit.json` and
|
||||
binds that object by checksum, size, and generation. Readers strictly reject
|
||||
unknown fields, version mismatches, pointer/commit identity mismatches, and
|
||||
objects that do not match their declaration.
|
||||
|
||||
The reader retains a temporary, clearly isolated compatibility path for a
|
||||
coherent legacy `current/manifest.json` plus `current/run_id.txt` pair. That
|
||||
path is removable after migration and is never used to write new state.
|
||||
|
||||
## Persistence Semantics
|
||||
|
||||
`manifest.LocalStore`:
|
||||
|
||||
- validates loaded documents;
|
||||
- normalizes missing maps/stage records;
|
||||
- writes atomically via temp file + rename;
|
||||
- writes through a sibling temporary file, syncing the completed file and
|
||||
destination directory after atomic replacement;
|
||||
- updates `updated_at` on save.
|
||||
|
||||
If the operating system or filesystem cannot sync a directory, save returns an
|
||||
explicit error instead of claiming crash-durable replacement. A returned error
|
||||
after the rename can therefore leave the new manifest visible but not confirmed
|
||||
durable; callers must reload it before retrying.
|
||||
|
||||
## Execution Semantics
|
||||
|
||||
The application runner marks an executing stage running and then succeeded or
|
||||
@@ -53,7 +159,10 @@ failed in both manifests, persisting each transition. On success it records
|
||||
outputs, logs, generated configuration references, and metadata. Artifact
|
||||
records may include optional contract and external provenance objects; old
|
||||
manifests remain compatible when those fields are absent. A successful forced
|
||||
rerun marks only succeeded downstream session-stage records stale.
|
||||
rerun marks only succeeded transitive dependent session-stage records stale.
|
||||
The application owns a fixed dependency relation distinct from execution order;
|
||||
dependents are returned in canonical order. Render and extract therefore never
|
||||
stale one another, while either can stale analyze, publish, and notify.
|
||||
|
||||
Starting an execution clears the current session-stage record's prior outputs,
|
||||
logs, generated configuration references, and metadata. Failed and skipped
|
||||
@@ -63,6 +172,12 @@ those details because resume validation and diagnosis may still require them
|
||||
before execution begins. Invocation run manifests remain immutable audit
|
||||
records of their own outcomes.
|
||||
|
||||
Aggregate lifecycle clearing deliberately preserves the analyze-owned
|
||||
per-artifact collection. This lets later reconciliation replace only evaluated
|
||||
entries without erasing unrelated current results. Other stages retain their
|
||||
existing aggregate-only lifecycle behavior and are forbidden from carrying the
|
||||
analyze-specific fields.
|
||||
|
||||
A stage may explicitly return a skipped disposition and stable reason. The
|
||||
runner persists that outcome in both manifests, clears older outputs for the
|
||||
session-stage record along with older logs, generated configuration references,
|
||||
@@ -74,26 +189,60 @@ cannot contain outputs.
|
||||
When an already-succeeded stage is skipped, the invocation run manifest records
|
||||
the `skip` action and reason. The session manifest deliberately retains its
|
||||
existing succeeded record because it remains the cross-invocation progress
|
||||
authority. Stages with a resume validator, currently extraction, may reject an
|
||||
otherwise eligible skip when the recorded durable result is obsolete; the
|
||||
runner marks it stale and executes it.
|
||||
authority. Extraction and analyze have resume validators and may reject an
|
||||
otherwise eligible skip when their selected durable evidence is obsolete; the
|
||||
runner marks the aggregate record stale and executes it. Analyze's validator
|
||||
can still accept a partial selection when only unrelated artifact records are
|
||||
stale.
|
||||
|
||||
Session manifest is the authoritative stage-progress ledger across invocations.
|
||||
Run manifest is invocation-scoped audit state.
|
||||
|
||||
Before an explicitly bounded execution starts after `prepare`, the application
|
||||
reads the session manifest and accepts only `succeeded` or `skipped` for every
|
||||
excluded canonical prefix stage. The first other status or absent record fails
|
||||
the request before layout mutation, adapter initialization, session-manifest
|
||||
writes, or run-manifest creation. Excluded prefix records are not passed to
|
||||
resume validators. Records after the selected end are not prerequisites and
|
||||
may be made stale by selected work without being scheduled.
|
||||
|
||||
After a publish commits remotely, any configured local cleanup is first recorded
|
||||
as a session-manifest obligation before deletion begins. Each target becomes
|
||||
complete only after its confined deletion (or safe absence check) and a
|
||||
successful manifest save. An incomplete obligation is retried when publish
|
||||
executes again and retains the committed run and remote identity that authorized
|
||||
it; an invocation that does not execute publish does not perform cleanup.
|
||||
|
||||
Each invocation derives campaign, session, run, local-path, and remote-prefix
|
||||
metadata from the validated resolved configuration as one projection. A persisted
|
||||
session manifest must agree on campaign and session identity before execution;
|
||||
the current projection is refreshed for every invocation while stage progress,
|
||||
inputs, and durable artifacts remain session history.
|
||||
|
||||
For handled failures after an invocation record is created, the runner records
|
||||
the failure on the session ledger and persists it before persisting the failed
|
||||
run audit record. This preserves the resume authority while making a partial
|
||||
persistence disagreement visible. Abrupt process death remains an accepted case
|
||||
where a durable running record can require operator interpretation.
|
||||
|
||||
## Invariants
|
||||
|
||||
- stage resume/skip decisions are session-manifest driven.
|
||||
- running, failed, and self-skipped stages do not retain result payloads from
|
||||
an earlier success.
|
||||
- stale stages retain prior details until replacement execution starts.
|
||||
- force reruns stale downstream succeeded stages.
|
||||
- force reruns stale succeeded stages in the fixed dependency relation.
|
||||
- run manifest does not replace session manifest as progress authority.
|
||||
- remote commitment is established by a verified current pointer and remote
|
||||
commit relationship, never by a mutable session-manifest boolean.
|
||||
|
||||
## Implementation And Tests
|
||||
|
||||
- Models and transitions: `internal/manifest/manifest.go`,
|
||||
`internal/manifest/run_manifest.go`
|
||||
- Remote commit model and readers: `internal/artifacts/remote_commit.go`,
|
||||
`internal/artifacts/current_state_commit.go`,
|
||||
`internal/artifacts/current_state_legacy.go`
|
||||
- Persistence and validation: `internal/manifest/store.go`
|
||||
- Package tests: `internal/manifest/*_test.go`
|
||||
- Assembled execution behavior: `internal/app/runner_test.go`,
|
||||
|
||||
@@ -35,7 +35,7 @@ progress and artifact services resolve durable inputs and outputs.
|
||||
| Artifacts and paths | `internal/artifacts`, `internal/pathsafe` | Artifact identities and resolution, local and remote path/key models, current-state discovery, and confined relative destinations. |
|
||||
| Previous-session cache | `internal/previouscache` | Deterministic planning and materialization requirements for configured previous-session inputs. |
|
||||
| Artifact policy | `internal/artifactpolicy` | Source and destination policy, configured artifact identity validation, and publish destination safety. |
|
||||
| Shared models and file operations | `internal/artifactmodel`, `internal/contracts`, `internal/fileops` | Transcript and artifact data contracts plus narrow atomic filesystem helpers. |
|
||||
| Shared models and file operations | `internal/artifactmodel`, `internal/contracts`, [`internal/fileops`](fileops.md) | Transcript and artifact data contracts plus durable single-file replacement helpers; unsupported directory syncing is reported explicitly. |
|
||||
| Logging | `internal/logging` | Application logger construction and shared structured logging behavior. |
|
||||
|
||||
The application boundary composes concrete implementations. Stages depend on
|
||||
@@ -43,6 +43,14 @@ Narratio-level contracts; external transport and SDK details remain in
|
||||
adapters. The normative rules for these relationships remain in
|
||||
[Architecture](../policy/architecture.md).
|
||||
|
||||
Pipeline execution and `session plan` share the same inclusive contiguous-range
|
||||
model. Planning clones session state and applies selected-stage transitions and
|
||||
resume validation in memory; it does not create invocation state or initialize
|
||||
stage-execution adapters. Command configuration loading can still retrieve a
|
||||
missing session file through configured remote storage. Analyze planning
|
||||
additionally exposes the artifact closure's targets, prerequisite rebuilds,
|
||||
execution order, and current reuse.
|
||||
|
||||
## Pipeline Stage Set
|
||||
|
||||
The implemented canonical order is:
|
||||
@@ -53,18 +61,25 @@ The implemented canonical order is:
|
||||
4. [`polish`](stage-polish.md)
|
||||
5. [`normalize`](stage-normalize.md)
|
||||
6. [`trim`](stage-trim.md)
|
||||
7. [`extract`](stage-extract.md)
|
||||
8. [`render`](stage-render.md)
|
||||
7. [`render`](stage-render.md)
|
||||
8. [`extract`](stage-extract.md)
|
||||
9. [`analyze`](stage-analyze.md)
|
||||
10. [`publish`](stage-publish.md)
|
||||
11. `notify` (placeholder)
|
||||
11. `notify` (no-op)
|
||||
|
||||
`notify` currently has optional notifier call behavior and no persisted pipeline
|
||||
outputs; its default collaborator is a no-op sender. The focused stage
|
||||
documents own implementation mechanics. The
|
||||
`notify` currently has no persisted pipeline outputs and uses the explicit
|
||||
`noop` notification mode. The focused stage documents own implementation
|
||||
mechanics. The
|
||||
[CLI](../cli.md) and [Operations](../operations.md) own user-visible invocation
|
||||
and execution semantics.
|
||||
|
||||
Execution order and invalidation are separate application contracts. The stage
|
||||
registry owns the flat execution sequence. The application orchestration owner
|
||||
uses a fixed, validated dependency relation to find transitive dependents in
|
||||
canonical order. In particular, `render` and `extract` are sibling consumers of
|
||||
trimmed transcript state: neither invalidates the other, while either can stale
|
||||
`analyze`, `publish`, and `notify`.
|
||||
|
||||
## Focused Documentation
|
||||
|
||||
- [Adapter Internals](adapters.md): external adapter boundaries, composition,
|
||||
@@ -84,8 +99,8 @@ and execution semantics.
|
||||
- [`polish`](stage-polish.md)
|
||||
- [`normalize`](stage-normalize.md)
|
||||
- [`trim`](stage-trim.md)
|
||||
- [`extract`](stage-extract.md)
|
||||
- [`render`](stage-render.md)
|
||||
- [`extract`](stage-extract.md)
|
||||
- [`analyze`](stage-analyze.md)
|
||||
- [`publish`](stage-publish.md)
|
||||
|
||||
|
||||
@@ -2,34 +2,124 @@
|
||||
|
||||
## Purpose
|
||||
|
||||
Execute selected configured Scriptorium artifacts in dependency order and materialize outputs.
|
||||
Reconcile configured Scriptorium artifacts, execute only required work in
|
||||
dependency order, and safely materialize validated outputs.
|
||||
|
||||
## Inputs
|
||||
|
||||
- configured artifacts from `pipeline.scriptorium.artifacts`
|
||||
- optional selected artifact keys supplied through the stage environment
|
||||
- built-in/configured/previous-session source references in artifact inputs
|
||||
- built-in, configured, extraction, and previous-session source references in
|
||||
artifact inputs
|
||||
|
||||
Supported source families:
|
||||
- built-ins: `narratio.transcript.*`, `narratio.bounds.session`
|
||||
- prepared stable inputs: `narratio.input.players`, `narratio.input.party`,
|
||||
`narratio.input.glossary`
|
||||
`narratio.input.glossary`, `narratio.input.spell_catalog`
|
||||
- configured artifacts: `narratio.artifact.<key>`
|
||||
- extraction lanes: `narratio.extraction.<key>`
|
||||
- previous-session cache: `narratio.previous_session.artifact.<key>`
|
||||
|
||||
## Outputs
|
||||
|
||||
- one materialized output per executed configured artifact (`output_path`)
|
||||
- one current per-artifact manifest record per validated materialized output
|
||||
- stage metadata describing selected/generated/reused artifacts
|
||||
|
||||
## Key Behavior
|
||||
|
||||
- skips with metadata when Scriptorium config is missing or no executable artifacts remain.
|
||||
- builds runtime artifact catalog (built-ins + configured artifacts).
|
||||
- marks non-executable configured artifacts as reusable when output files already exist.
|
||||
- when `pipeline.scriptorium` is absent or no configured artifact is
|
||||
executable, completes successfully with no outputs and records explanatory
|
||||
metadata. This is not an explicit self-skip: both manifests record success,
|
||||
satisfy publish's prerequisite, and an ordinary later run reuses the result
|
||||
while the effective set remains empty. Enabling or selecting an artifact
|
||||
later makes missing versioned evidence non-resumable and schedules it without
|
||||
requiring force.
|
||||
- builds a runtime artifact catalog containing built-ins, configured artifacts,
|
||||
and configured extraction lanes. Extraction availability is hydrated only
|
||||
from compatible successful extraction evidence.
|
||||
- uses enabled configured artifacts by default. An explicit `--artifacts`
|
||||
selection is a one-invocation override that makes exactly the named
|
||||
configured artifacts explicit targets even when disabled. The work planner
|
||||
adds required configured prerequisites, reuses current ones, and schedules
|
||||
stale, missing, or otherwise non-current prerequisites before dependents.
|
||||
- makes a non-executable configured artifact reusable only when its current
|
||||
manifest record and durable output pass the configured-artifact evidence
|
||||
contract; an incidental or stale canonical file is unavailable.
|
||||
- validates selected artifact dependency order (cycle-safe topo ordering).
|
||||
- resolves required/optional inputs per artifact source definition.
|
||||
- resolves prepared stable input sources from `inputs/*.yml` materialized by `prepare`.
|
||||
- resolves required/optional inputs per artifact source definition into an
|
||||
ordered semantic identity. Each identity records the configured input name,
|
||||
canonical source ID, required policy, explicit presence, source contract,
|
||||
checksum, size, and a source-based logical identity. Workspace paths and
|
||||
producer run IDs are excluded.
|
||||
- orders input identities by configured input name independently of Go map
|
||||
iteration. Runtime adapter paths remain a separate execution-only map.
|
||||
- omits an unavailable optional input from the adapter request while retaining
|
||||
explicit absence in its semantic identity; an unavailable required input
|
||||
fails.
|
||||
- resolves prepared stable input sources through the shared manifest-authoritative
|
||||
identity resolver; it does not accept incidental files or fall back to
|
||||
campaign/session source paths.
|
||||
- reuses checksums and sizes from validated prepared, extraction, and current
|
||||
configured-artifact evidence. Other resolved inputs are hashed as confined
|
||||
regular files with streaming reads and the central resolved-artifact size
|
||||
limit.
|
||||
- owns a versioned SHA-256 fingerprint contract with one fixed-field canonical
|
||||
JSON payload and no map serialization. Configured artifacts are fingerprinted
|
||||
in deterministic dependency order.
|
||||
- fingerprints the normalized artifact key, prompt and profile identifiers,
|
||||
effective render-debug behavior, session-relative output identity, sorted
|
||||
dependency keys, ordered input declarations and semantic identities,
|
||||
validated current dependency-output identities, and sorted effective
|
||||
Scriptorium variables (including Narratio's sticky session variable).
|
||||
- provides read-only reconciliation that classifies each configured record as
|
||||
current, stale, missing, failed, legacy, or otherwise non-resumable, and
|
||||
separately identifies manifest records removed from current configuration.
|
||||
A record is current only when its fingerprint version and value match and its
|
||||
configured output still passes manifest-authoritative evidence validation.
|
||||
- owns a read-only typed work planner. Its explicit targets are enabled
|
||||
artifacts by default or the exact normalized `--artifacts` selection when
|
||||
supplied. It closes targets over configured prerequisites, orders the closure
|
||||
topologically, reuses current members, and schedules every non-current member
|
||||
before its dependents.
|
||||
- force applies only to explicit targets. A current prerequisite is reused
|
||||
unless it is itself an explicit forced target; disabled prerequisites may be
|
||||
rebuilt when required, while unrelated disabled artifacts are excluded.
|
||||
- the work plan carries explicit targets, prerequisite-only work, deterministic
|
||||
execution and reuse lists, invalidated and removed records, and a cloned
|
||||
projected record collection. Valid unrelated configured records survive the
|
||||
projection, removed records are omitted, and legacy files never become
|
||||
current without regeneration.
|
||||
- implements aggregate resume validation by running the same read-only catalog,
|
||||
fingerprint reconciliation, and work planner used by execution. A succeeded
|
||||
aggregate record is reusable exactly when the selected closure schedules no
|
||||
artifact work; stale unrelated records do not block a partial selection.
|
||||
- exposes the typed artifact decision to `session plan`. Planning applies it to
|
||||
a cloned manifest after modeling earlier selected stage transitions, so
|
||||
aggregate run/skip and artifact execute/reuse decisions match the ordinary
|
||||
runner without creating durable state or invoking Scriptorium.
|
||||
- executes only the work plan's scheduled entries. Manifest-validated current
|
||||
prerequisites remain available through the runtime catalog without invoking
|
||||
Scriptorium; newly produced prerequisites enter that catalog with the same
|
||||
contract, checksum, and size identity used for persisted current evidence.
|
||||
- keeps adapter output in the invocation's run-local analyze directory until
|
||||
it is a safe, non-empty, bounded regular file with a calculated checksum and
|
||||
complete output contract. Canonical replacement uses the shared atomic file
|
||||
operation boundary and verifies that the installed checksum matches the
|
||||
validated run-local bytes.
|
||||
- records each successful artifact's freshly computed fingerprint, canonical
|
||||
relative output path, contract, checksum, size, producer run ID, bounded
|
||||
Scriptorium provenance, logs, and generated configuration references in the
|
||||
analyze-owned projection.
|
||||
- preserves valid unrelated current records during partial execution. If a
|
||||
rebuilt output's bytes and contract are unchanged, unselected dependents may
|
||||
remain current. If that semantic identity changes, unselected transitive
|
||||
dependents become stale without being executed; dependents included in the
|
||||
invocation are evaluated in dependency order instead.
|
||||
- reports all evaluated targets and prerequisites in invocation state. The
|
||||
runner reconstructs aggregate session outputs from every current session
|
||||
record and invocation outputs from only records produced by the current run.
|
||||
Unrelated stale records do not make an otherwise successful partial
|
||||
invocation fail.
|
||||
- resolves previous-session sources from local `previous/` cache only.
|
||||
- runs optional render-debug, then artifact execution.
|
||||
- validates non-empty output files and materializes canonical outputs.
|
||||
@@ -44,11 +134,33 @@ Supported source families:
|
||||
guidance.
|
||||
- dependency cycles or unavailable required dependencies fail.
|
||||
- adapter validation failures fail stage.
|
||||
- a scheduled artifact failure returns the restricted analyze-state projection
|
||||
with the active artifact marked `failed`, a bounded error, and no output
|
||||
authority. Current transitive dependents become stale without execution.
|
||||
- earlier artifacts from the invocation remain current only after their
|
||||
run-local output passed validation and canonical materialization. They remain
|
||||
in invocation history; unattempted later artifacts do not appear there.
|
||||
- unrelated current records survive a partial failure. Old canonical bytes for
|
||||
the failed artifact and newly materialized bytes whose projection cannot be
|
||||
persisted are incidental, not current evidence.
|
||||
- the runner persists a valid partial projection before it marks aggregate
|
||||
analyze failed and invalidates publish and notify through the application
|
||||
dependency relation. Projection-persistence errors retain the last durable
|
||||
per-artifact authority and are joined with the original failure context.
|
||||
|
||||
## Invariants
|
||||
|
||||
- `analyze` performs no remote storage calls for previous-session source resolution.
|
||||
- input-identity resolution is read-only: it does not invoke adapters,
|
||||
materialize outputs, update status, or create run records.
|
||||
- fingerprints exclude timeouts, retries, timestamps, producer and Narratio run
|
||||
IDs, executable and config paths, workspace roots, diagnostic locations, and
|
||||
executable or private transitive configuration contents. A change that is
|
||||
visible only inside Scriptorium—such as a file privately loaded by its config
|
||||
path—requires an explicit forced regeneration.
|
||||
- output provenance and metadata are deterministic per execution.
|
||||
- a canonical file without current per-artifact manifest evidence is never
|
||||
promoted to current state.
|
||||
|
||||
## Related Contracts And Tests
|
||||
|
||||
@@ -57,4 +169,13 @@ Supported source families:
|
||||
- [CLI](../cli.md) owns user-visible artifact selection.
|
||||
- [Scriptorium](../integrations/scriptorium.md) owns the subprocess contract.
|
||||
- Implementation and tests: `internal/stage/analyze.go`,
|
||||
`internal/stage/analyze_test.go`
|
||||
`internal/stage/analyze_input_identity.go`, `internal/stage/analyze_test.go`,
|
||||
`internal/stage/analyze_input_identity_test.go`,
|
||||
`internal/stage/analyze_fingerprint.go`,
|
||||
`internal/stage/analyze_fingerprint_test.go`,
|
||||
`internal/stage/analyze_reconciliation.go`, and
|
||||
`internal/stage/analyze_reconciliation_test.go`,
|
||||
`internal/stage/analyze_plan.go`, `internal/stage/analyze_plan_test.go`, and
|
||||
`internal/stage/analyze_incremental_execution_test.go`, and
|
||||
`internal/stage/analyze_failure_test.go`,
|
||||
`internal/stage/analyze_resume.go`, and `internal/stage/analyze_resume_test.go`
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
## Responsibility
|
||||
|
||||
`extract` runs after `trim` and before `render`. It converts the canonical
|
||||
`extract` runs after `render` and before `analyze`. It converts the canonical
|
||||
`narratio.transcript.final_trimmed` JSON into configured Notarius lane artifacts.
|
||||
An omitted or disabled Notarius section makes the stage explicitly self-skip
|
||||
with reason `notarius_disabled`, no outputs, and no Notarius runner.
|
||||
@@ -17,33 +17,57 @@ procedures belong in [Operations](../operations.md).
|
||||
`internal/stage/extract.go`:
|
||||
|
||||
1. resolves the final trimmed transcript from the shared artifact catalog;
|
||||
2. resolves and fingerprints the Notarius invocation contract;
|
||||
3. creates a run-local staging directory and invokes the injected
|
||||
2. resolves every configured prepared reference through the shared
|
||||
manifest-authoritative identity resolver before creating run-local output;
|
||||
3. streams each verified reference into an invocation-local snapshot and
|
||||
rejects any source change observed while copying;
|
||||
4. fingerprints the Notarius invocation contract, including sorted reference
|
||||
identities;
|
||||
5. creates a run-local staging directory and invokes the injected
|
||||
`notarius.Runner`;
|
||||
4. validates the successful receipt, confined index, configured required lane
|
||||
descriptors, and regular payload files;
|
||||
5. atomically promotes the complete bundle to its immutable durable location;
|
||||
6. records one non-selectable `notarius_index` output and one selectable
|
||||
6. revalidates the reference snapshots, then validates the v2 successful
|
||||
receipt, confined index, management documents, configured required lane
|
||||
descriptors, validation summaries, and regular payload files;
|
||||
7. atomically promotes the complete bundle to its immutable durable location;
|
||||
8. records one non-selectable `notarius_index` output and one selectable
|
||||
`notarius_lane` output per configured lane; and
|
||||
7. registers each lane as `narratio.extraction.<output_key>` for downstream
|
||||
9. registers each lane as `narratio.extraction.<output_key>` for downstream
|
||||
Scriptorium and publish resolution.
|
||||
|
||||
Lane records retain checksum, contract, producer run ID, and Notarius system,
|
||||
run, pipeline, and lane provenance. Stage metadata retains the durable bundle
|
||||
root, receipt, diagnostic paths, rejection/warning summaries, producing
|
||||
Narratio run ID, and invocation fingerprint. Validation completes before
|
||||
Narratio run ID, the resolved trimmed-input identity, and invocation
|
||||
fingerprint. The input identity binds the exact transcript bytes, canonical
|
||||
source ID, producer stage/output/run identity, and resolution provenance.
|
||||
Reference metadata contains only selector, source ID, canonical session-relative
|
||||
path, checksum, and size; adapter requests receive selector and absolute
|
||||
invocation-local snapshot path, never payload contents. Snapshot bytes must
|
||||
match the prepared identity both before and after Notarius runs, so a concurrent
|
||||
prepared-file replacement cannot make recorded provenance describe different
|
||||
bytes from those supplied to Notarius.
|
||||
Validation completes before
|
||||
promotion, so a rejected result cannot expose a partial durable bundle.
|
||||
|
||||
Any executed extraction outcome that replaces a different effective outcome
|
||||
marks succeeded downstream stages stale. Repeating the same disabled self-skip
|
||||
with no outputs is stable and does not repeatedly invalidate downstream stages.
|
||||
marks succeeded analysis and delivery dependents stale. Render is an independent
|
||||
sibling and remains current. Repeating the same disabled self-skip with no
|
||||
outputs is stable and does not repeatedly invalidate dependent stages.
|
||||
|
||||
## Resume Validation
|
||||
|
||||
`internal/stage/extract_resume.go` permits a skip only when the existing stage
|
||||
record succeeded and still matches the current invocation fingerprint. The
|
||||
fingerprint covers the resolved executable and config paths, pipeline ID,
|
||||
timeout, working directory, and sorted configured output contracts.
|
||||
timeout, working directory, sorted configured output contracts, the current
|
||||
direct trimmed-transcript identity, and sorted prepared-reference identities.
|
||||
The same reference helper and transcript identity are resolved again for
|
||||
artifact evidence, so changing the current transcript bytes or producer
|
||||
identity makes the prior extraction obsolete.
|
||||
|
||||
A valid prepared-reference change makes extraction non-resumable. Missing,
|
||||
unsafe, or checksum-inconsistent prepared evidence is a hard validation error
|
||||
with prepare-force guidance because an immediate extract rerun cannot succeed.
|
||||
|
||||
The validator then checks the producing run identity, canonical immutable
|
||||
bundle root, path confinement and absence of symlink components, receipt
|
||||
@@ -59,8 +83,10 @@ Operators must force extraction after changing any such input.
|
||||
## Failure Behavior
|
||||
|
||||
Adapter startup, timeout, nonzero exit, receipt decoding, path confinement,
|
||||
index compatibility, required-lane rejection, payload inspection, checksum, or
|
||||
promotion errors fail the stage through ordinary manifest transition handling.
|
||||
index compatibility, inconsistent warning or diagnostic envelopes,
|
||||
required-lane rejection or incomplete validation, payload inspection,
|
||||
checksum, or promotion errors fail the stage through ordinary manifest
|
||||
transition handling.
|
||||
Stdout receipt and stderr diagnostics remain separate. Downstream stages are
|
||||
not given selectable extraction sources unless the complete configured result
|
||||
has passed validation and promotion.
|
||||
|
||||
@@ -17,7 +17,8 @@ Run Audita polishing on base transcript and produce polished transcript.
|
||||
## Key Behavior
|
||||
|
||||
- resolves base transcript from merge outputs/canonical fallback.
|
||||
- invokes Audita with configured model/module/runtime options.
|
||||
- invokes an Audita runner configured with static model/runtime options; the
|
||||
invocation supplies paths and modules.
|
||||
- validates processed transcript structure (`segments` array required).
|
||||
- validates optional report JSON.
|
||||
- materializes canonical outputs; records logs/generated config and adapter metadata.
|
||||
|
||||
@@ -8,6 +8,7 @@ Materialize canonical current-session inputs before processing stages.
|
||||
|
||||
- resolved campaign, session, and pipeline configuration
|
||||
- stable input files (`speakers`, `autocorrect`, `glossary`, `players`, `party`)
|
||||
- optional spell-catalog overlay
|
||||
- one resolved local or S3 audio source
|
||||
- enabled configured artifact input requirements for previous-session sources
|
||||
|
||||
@@ -21,6 +22,7 @@ Materialize canonical current-session inputs before processing stages.
|
||||
- `inputs/glossary.yml`
|
||||
- `inputs/players.yml`
|
||||
- `inputs/party.yml`
|
||||
- optional `inputs/spell_catalog.json`
|
||||
- `audio/*.flac`
|
||||
- optional `previous/manifest.json`
|
||||
- optional `previous/artifacts/**`
|
||||
@@ -30,21 +32,29 @@ Materialize canonical current-session inputs before processing stages.
|
||||
|
||||
- validates required config/store state.
|
||||
- enforces local audio vs S3 audio mutual exclusivity.
|
||||
- rejects duplicate explicit local audio sources after resolution.
|
||||
- gives distinct local source paths with the same basename deterministic unique
|
||||
prepared filenames so neither source is overwritten.
|
||||
- materializes S3 audio through spool/cache-aware logic.
|
||||
- materializes a configured spell catalog with checksum and provenance, or
|
||||
safely removes an obsolete canonical spell catalog and its manifest record
|
||||
when the effective input is omitted.
|
||||
- scans enabled configured artifact inputs for `narratio.previous_session.artifact.*` requirements.
|
||||
- when previous requirements exist:
|
||||
- clears managed `previous/` state;
|
||||
- builds previous-cache remote plan;
|
||||
- clears managed `previous/` state on every invocation, then, when requirements exist:
|
||||
- resolves the pointer-selected previous source through the shared resolver;
|
||||
- downloads previous manifest/artifacts;
|
||||
- records previous inputs in `manifest.inputs`.
|
||||
|
||||
Required previous-session inputs fail when unavailable; optional missing inputs are skipped.
|
||||
Required previous-session inputs fail when unavailable; optional missing inputs
|
||||
are typed skipped results. Committed sources use their exact source-to-destination
|
||||
mapping, while the isolated legacy reader rejects ambiguous fallback matches.
|
||||
|
||||
## Invariants
|
||||
|
||||
- only `prepare` hydrates canonical `previous/` cache state.
|
||||
- managed previous artifacts are stored under `previous/artifacts/**` without
|
||||
duplicate `artifacts/artifacts/` nesting.
|
||||
- managed `previous/` state represents only the current requirement set.
|
||||
- `manifest.inputs` ordering is deterministic (`kind`, `path`).
|
||||
|
||||
## Related Contracts And Tests
|
||||
|
||||
@@ -9,34 +9,58 @@ Upload run/session outputs to object storage and atomically advance remote curre
|
||||
- successful preceding stages from the [canonical stage set](overview.md#pipeline-stage-set)
|
||||
- invocation-scoped run files
|
||||
- resolved publish output rules
|
||||
- effective publish locks (static + remote merged lock set)
|
||||
- effective publish locks (static + remote merged lock set), revalidated at the
|
||||
remote commit boundary
|
||||
- durable previous-session cache files when present
|
||||
|
||||
## Outputs
|
||||
|
||||
- uploaded invocation record and selected publish outputs;
|
||||
- uploaded durable previous-session cache files when present;
|
||||
- updated remote current manifest; and
|
||||
- remote current-run commit marker, written last.
|
||||
- immutable run-scoped commit manifest; and
|
||||
- current commit pointer, written last.
|
||||
|
||||
Exact remote placement and the operator workflow belong in
|
||||
[Operations](../operations.md#publish-workflow).
|
||||
|
||||
## Key Behavior
|
||||
|
||||
- stage can self-skip when publish disabled or run upload disabled.
|
||||
- when publishing or run upload is disabled, completes successfully with no
|
||||
outputs and records explanatory metadata. This is not an explicit self-skip:
|
||||
both manifests record success, and an ordinary later run reuses that result
|
||||
until publish is forced.
|
||||
- validates prerequisite stage success and object-store availability.
|
||||
- collects a deterministic run file list plus run `manifest.json`, excluding
|
||||
`audio/**` and the run-local `extract/notarius-output/**` staging bundle.
|
||||
- keeps run-local Notarius receipt and stderr diagnostics eligible for the run
|
||||
archive.
|
||||
- resolves publish output sources through runtime artifact catalog and manifest-aware resolution.
|
||||
- derives a deterministic run-archive allowlist from the validated run
|
||||
`manifest.json`: declared run-local outputs, logs, generated configs, and the
|
||||
manifest itself. Unlisted workspace files are not archive candidates.
|
||||
- opens each archive candidate beneath its archive root without following
|
||||
symlinked ancestors or leaf entries, verifies that it is a regular file and
|
||||
checks a declared checksum when present, then streams the opened descriptor.
|
||||
- derives the durable previous-cache archive from its validated manifest using
|
||||
the same confinement and regular-file checks.
|
||||
- resolves publish output sources through runtime artifact catalog and
|
||||
manifest-aware resolution. Configured Scriptorium outputs are publishable
|
||||
only from validated `current` per-artifact analyze evidence; an incidental
|
||||
canonical file, legacy aggregate output, stale/failed/unselected record, or
|
||||
mismatched path, size, or checksum remains unavailable. This does not change
|
||||
the explicit compatibility policies owned by built-in, extraction, or
|
||||
previous-session sources.
|
||||
- publishes extraction lanes only through explicit configured output rules;
|
||||
neither run-local nor durable Notarius bundles are scanned or uploaded wholesale.
|
||||
- selected artifact filter applies to configured artifact sources only.
|
||||
- locked outputs are skipped intentionally (including required ones).
|
||||
- optional missing outputs are skipped; required missing unlocked outputs fail.
|
||||
- writes remote current manifest before current run pointer.
|
||||
- creates one complete immutable source-to-destination mapping before upload;
|
||||
- uploads and verifies every declared immutable object and the commit manifest;
|
||||
- updates `current/commit-pointer.json` exactly once, last; and
|
||||
- does not write the legacy `current/manifest.json` or `current/run_id.txt` pair.
|
||||
- rechecks remote lock state immediately before the pointer update. A newly
|
||||
committed lock aborts selection, leaving any uploaded immutable attempt
|
||||
unselected.
|
||||
- reads the mutable remote lock document through a direct limit-plus-one read
|
||||
capped by `MaxRemoteLockStoreBytes` (1 MiB), retaining the generation returned
|
||||
with the opened body for conditional updates. Oversized lock documents fail
|
||||
before YAML decoding; published artifact payloads do not use this limit.
|
||||
|
||||
## Metadata Signals
|
||||
|
||||
@@ -47,16 +71,23 @@ Includes counts/lists for:
|
||||
- skipped optional outputs
|
||||
- skipped unselected outputs
|
||||
- locked outputs
|
||||
- current-state key paths
|
||||
- `current_pointer_written`
|
||||
- remote commit and current-pointer key paths
|
||||
- the run identifier selected by the commit
|
||||
|
||||
## Invariants
|
||||
|
||||
- `current/run_id.txt` is the remote commit marker and is written last.
|
||||
- run upload excludes `audio/**` and `extract/notarius-output/**`.
|
||||
- `extract/notarius.receipt.json` and `extract/notarius.stderr.log` remain
|
||||
eligible run-record diagnostics.
|
||||
- publish locks are not overridden by `--force`.
|
||||
- `current/commit-pointer.json` is the remote commit marker and is written last.
|
||||
- run files, selected outputs, previous-cache files, and the committed session
|
||||
manifest are all declared by an immutable commit under the run prefix.
|
||||
- run and previous uploads contain only manifest-declared regular files opened
|
||||
from verified descriptors; symlinks, special files, replacement races, and
|
||||
undeclared entries are rejected or ignored before uploads begin.
|
||||
- run-local diagnostics, including Notarius receipt and stderr files, are
|
||||
archived only when recorded by the run manifest.
|
||||
- publish locks are not overridden by `--force`; remote locks are revalidated
|
||||
immediately before current-state selection.
|
||||
- post-commit local cleanup is authorized by the committed publish metadata and
|
||||
is durably recorded by the application lifecycle before any local deletion.
|
||||
|
||||
The commit boundary and cleanup gate are normative architecture invariants; see
|
||||
[Architecture](../policy/architecture.md#publish-commit-boundary).
|
||||
|
||||
@@ -3,6 +3,10 @@
|
||||
## Purpose
|
||||
|
||||
Render Markdown transcript artifacts from normalized JSON transcripts via Seriatim.
|
||||
It runs after `trim` and before `extract` in the canonical sequence. Render and
|
||||
extract are independent sibling consumers: replacing render output does not
|
||||
invalidate extraction, but it does invalidate succeeded analysis and delivery
|
||||
records that may consume rendered transcripts.
|
||||
|
||||
## Inputs
|
||||
|
||||
@@ -20,7 +24,9 @@ Render Markdown transcript artifacts from normalized JSON transcripts via Seriat
|
||||
- resolves inputs manifest-first, then canonical fallback.
|
||||
- writes run-local outputs first, then materializes canonical session outputs.
|
||||
- records input provenance, output paths, adapter metadata, logs, and generated config refs.
|
||||
- skips with stage metadata when `pipeline.render.enabled=false`.
|
||||
- when `pipeline.render.enabled=false`, completes successfully with no outputs
|
||||
and records explanatory metadata. This is not an explicit self-skip: both
|
||||
manifests record success, and enabling render later requires a forced run.
|
||||
|
||||
## Failure Semantics
|
||||
|
||||
|
||||
@@ -15,16 +15,21 @@ Generate raw per-speaker transcripts from prepared audio using WhisperX.
|
||||
## Key Behavior
|
||||
|
||||
- discovers prepared audio from manifest inputs or canonical audio directory.
|
||||
- derives speaker ID from `.flac` basename.
|
||||
- derives the transcript identity from the prepared `.flac` filename.
|
||||
- dispatches WhisperX requests through a bounded worker pool.
|
||||
- validates each output as JSON.
|
||||
- writes run-local outputs then materializes canonical transcript outputs.
|
||||
- writes run-local outputs then materializes canonical transcript outputs only
|
||||
after every planned request succeeds.
|
||||
|
||||
## Invariants
|
||||
|
||||
- speaker basenames must be unique.
|
||||
- prepared audio identities must be unique; prepare disambiguates distinct
|
||||
source paths that share a basename.
|
||||
- output path returned by adapter must match requested output path.
|
||||
- each successful output is validated before stage success.
|
||||
- an empty adapter result path means the requested path; adapters cannot select
|
||||
an alternate destination.
|
||||
- each successful output is validated before stage success, and cancellation or
|
||||
incomplete dispatch cannot be reported as a successful result.
|
||||
|
||||
## Related Contracts And Tests
|
||||
|
||||
|
||||
@@ -12,14 +12,23 @@ operator-selected storage fields and credential mechanisms belong in
|
||||
`storage.ObjectStore` interface:
|
||||
|
||||
- `List(ctx, prefix)`
|
||||
- `Read(ctx, key)` returns an object body and the generation observed with it
|
||||
- `Download(ctx, key, localPath)`
|
||||
- `Upload(ctx, localPath, key, opts)`
|
||||
- `UploadConditional(ctx, source, key, opts, condition)`
|
||||
- `Exists(ctx, key)`
|
||||
|
||||
Key invariant:
|
||||
- callers pass full bucket-relative keys;
|
||||
- storage implementations do not infer campaign/session/run prefixes.
|
||||
|
||||
`ReadObjectBounded` is the shared mechanism for small control objects. It opens
|
||||
one object version, returns the metadata observed with that body, rejects an
|
||||
oversized known size before transfer, and still performs a context-aware
|
||||
limit-plus-one read. It closes the body on every exit. Callers own the policy
|
||||
limit and add the control-object category to errors; this helper is not used for
|
||||
large artifact payloads.
|
||||
|
||||
## Composition
|
||||
|
||||
`NewObjectStoreFromConfig` constructs the S3-backed implementation from
|
||||
@@ -31,13 +40,20 @@ not own discovery, defaults, or configuration validation.
|
||||
|
||||
- normalizes object keys.
|
||||
- `List` paginates and returns normalized `ObjectInfo`.
|
||||
- A truncated S3 listing must supply a new, non-empty continuation token;
|
||||
otherwise listing fails with bucket and prefix context instead of looping.
|
||||
- `Download` writes local files with parent directory creation.
|
||||
- `Upload` streams local file and returns remote metadata.
|
||||
- `Read` binds a returned body to its S3 ETag. `UploadConditional` maps an ETag
|
||||
match or absence precondition directly to the provider request and reports a
|
||||
failed precondition without performing a local check-then-write replacement.
|
||||
- `Exists` maps not-found responses to `false`.
|
||||
|
||||
## Invariants
|
||||
|
||||
- storage layer is stateless regarding manifest/stage progression.
|
||||
- bounded reads never retain more than the caller's limit plus one byte and do
|
||||
not replace owner-specific size policy.
|
||||
- publish ordering semantics are owned by stage/app code, not storage adapters.
|
||||
|
||||
## Implementation And Tests
|
||||
|
||||
@@ -13,8 +13,16 @@ previous-cache path construction. `SessionPathsFor` provides the session-scoped
|
||||
path model, and layout creation goes through `EnsureLayoutFor`. Callers should
|
||||
consume those helpers instead of rebuilding relative paths.
|
||||
|
||||
`internal/pathsafe` and application cleanup helpers enforce confinement for
|
||||
relative destinations and deletion targets.
|
||||
`internal/pathsafe` validates relative destinations. `internal/fileops` opens
|
||||
cleanup roots and their descendants through no-follow directory handles before
|
||||
removing them.
|
||||
|
||||
`internal/fileops` owns the ordinary workspace mode contract. On POSIX,
|
||||
`WorkspaceDirectoryMode` is setgid `02775` and `WorkspaceFileMode` is `0664`.
|
||||
`EnsureWorkspaceDirectory` reapplies the directory mode after creation so a
|
||||
restrictive umask cannot remove group access, while retaining existing ownership
|
||||
and group. Credential paths are outside this contract; the platform-specific
|
||||
operational requirements are in [Operations](../operations.md#workspace-permissions).
|
||||
|
||||
## Run-Local Stage Layout
|
||||
|
||||
@@ -36,21 +44,30 @@ existing destination. Exact physical paths belong in
|
||||
|
||||
## Locking
|
||||
|
||||
`artifacts.LocalStore` enforces the single-writer session lock via `.lock`
|
||||
(`ErrLockConflict` on contention).
|
||||
`artifacts.LocalStore` enforces the single-writer session lock via an
|
||||
operating-system lock held on `.lock` (`ErrLockConflict` on contention). The
|
||||
file retains owner metadata after release or process death; its existence is
|
||||
not evidence that a lock is active. Command and restore flows wait for this
|
||||
lock only while their context remains active, and report a release failure.
|
||||
|
||||
## Cleanup Semantics
|
||||
|
||||
Automatic post-publish cleanup:
|
||||
|
||||
- only runs when publish actually executed and succeeded;
|
||||
- requires `uploaded=true` and `current_pointer_written=true` metadata;
|
||||
- is created only after a successful publish commit with complete publish
|
||||
metadata, then is persisted before any deletion;
|
||||
- requires `uploaded=true`, a remote commit key, and a current commit-pointer
|
||||
key in publish metadata;
|
||||
- consumes the resolved cleanup policy described in
|
||||
[Configuration](../config.md);
|
||||
- refuses unsafe deletes (root delete, out-of-root delete, symlink paths).
|
||||
- refuses unsafe deletes (root delete, out-of-root delete, and symlinked
|
||||
ancestors or entries);
|
||||
- retries any recorded incomplete target on later invocations even when no
|
||||
publish work is selected. Missing targets are a successful, idempotent
|
||||
cleanup result only after the completion evidence is saved.
|
||||
|
||||
Manual cleanup uses the same scoped-target checks. Invocation syntax and exact
|
||||
deletion scope belong in [CLI](../cli.md#clean) and
|
||||
Manual cleanup uses the same root-confined deletion mechanism. Invocation
|
||||
syntax and exact deletion scope belong in [CLI](../cli.md#clean) and
|
||||
[Operations](../operations.md#cleanup).
|
||||
|
||||
## Invariants
|
||||
@@ -58,6 +75,8 @@ deletion scope belong in [CLI](../cli.md#clean) and
|
||||
- campaign-aware session root is mandatory.
|
||||
- manifest-driven stage state is durable across runs.
|
||||
- cleanup guardrails prevent destructive root/out-of-scope deletion.
|
||||
- ordinary workspace paths retain group-writable directory and file modes across
|
||||
nested creation, replacement, and Notarius promotion.
|
||||
|
||||
## Implementation And Tests
|
||||
|
||||
@@ -65,10 +84,11 @@ deletion scope belong in [CLI](../cli.md#clean) and
|
||||
`internal/artifacts/local.go`
|
||||
- Run-local materialization: `internal/stage/run_local.go`
|
||||
- Immutable bundle promotion: `internal/fileops/directory.go`
|
||||
- Cleanup confinement: `internal/app/cleanup_targets.go`,
|
||||
`internal/app/post_publish_cleanup.go`
|
||||
- Workspace modes: `internal/fileops/modes.go`
|
||||
- Cleanup confinement: `internal/fileops/cleanup.go`,
|
||||
`internal/app/cleanup_targets.go`, `internal/app/post_publish_cleanup.go`
|
||||
- Tests: `internal/artifacts/paths_model_test.go`,
|
||||
`internal/artifacts/local_test.go`, `internal/stage/run_local_test.go`,
|
||||
`internal/fileops/directory_test.go`,
|
||||
`internal/app/cleanup_targets_test.go`,
|
||||
`internal/fileops/directory_test.go`, `internal/fileops/modes_posix_test.go`,
|
||||
`internal/fileops/cleanup_test.go`, `internal/app/cleanup_targets_test.go`,
|
||||
`internal/app/post_publish_cleanup_test.go`
|
||||
|
||||
@@ -36,7 +36,12 @@ narratio session init 2026-04-04 --remote --force
|
||||
|
||||
If `campaign.yml` sets `session_template_file`, `session init` renders it. Template variables must resolve to concrete values.
|
||||
|
||||
Campaigns must provide stable input files for speakers, autocorrect, glossary, players, and party. Session files may override those paths for one session. The `prepare` stage materializes them under `inputs/`; configured Scriptorium artifacts can reference prepared `players`, `party`, and `glossary` files with `narratio.input.players`, `narratio.input.party`, and `narratio.input.glossary`.
|
||||
Campaigns must provide stable input files for speakers, autocorrect, glossary,
|
||||
players, and party, and may provide an optional spell-catalog overlay. Session
|
||||
files may override those paths for one session. The `prepare` stage
|
||||
materializes them under `inputs/`; configured consumers use the prepared files,
|
||||
never the original campaign or session source paths. Field definitions and
|
||||
source IDs are in [Configuration](./config.md#notarius-reference-bindings).
|
||||
|
||||
## Standard Session Workflow
|
||||
|
||||
@@ -75,8 +80,8 @@ Canonical stage order:
|
||||
4. `polish`
|
||||
5. `normalize`
|
||||
6. `trim`
|
||||
7. `extract`
|
||||
8. `render`
|
||||
7. `render`
|
||||
8. `extract`
|
||||
9. `analyze`
|
||||
10. `publish`
|
||||
11. `notify`
|
||||
@@ -85,12 +90,20 @@ Execution rules:
|
||||
|
||||
- succeeded stages are skipped unless `--force` is set;
|
||||
- `run` continues interrupted or partially completed sessions by running non-succeeded stages;
|
||||
- forcing an upstream stage marks succeeded downstream stages as `stale` before
|
||||
the replacement runs; and
|
||||
- forcing a stage marks succeeded transitive dependents as `stale` before the
|
||||
replacement runs; render and extract are independent siblings; and
|
||||
- an executed failure, changed self-skip, or success that replaces a different
|
||||
effective upstream outcome also marks succeeded downstream stages stale. A
|
||||
effective outcome uses the same fixed dependency relation. A
|
||||
repeated self-skip with the same reason and no outputs is stable and does not
|
||||
perpetually rerun downstream work.
|
||||
perpetually rerun dependent work.
|
||||
|
||||
An explicit self-skip is a durable `skipped` stage outcome that later runs
|
||||
reconsider. It differs from successful no-output execution: disabled `render`
|
||||
and `publish`, and absent or no-executable `analyze`, record `succeeded` with
|
||||
metadata and no outputs. Ordinary later runs reuse those successful results;
|
||||
force the affected stage after enabling or configuring it. Optional artifact
|
||||
inputs are omitted only from the consuming artifact invocation and do not make
|
||||
the stage self-skip.
|
||||
|
||||
Single-stage execution:
|
||||
|
||||
@@ -98,14 +111,81 @@ Single-stage execution:
|
||||
narratio run-stage normalize 2026-04-04 --force
|
||||
```
|
||||
|
||||
Contiguous bounded execution uses inclusive canonical endpoints:
|
||||
|
||||
```bash
|
||||
narratio session plan 2026-04-04 --from extract --through analyze --force
|
||||
narratio run 2026-04-04 --from extract --through analyze --force
|
||||
```
|
||||
|
||||
Omitting `--from` selects from `prepare`; omitting `--through` selects through
|
||||
`notify`. Force applies only within the selected range. Repeating `--from`,
|
||||
`--through`, or `--force` is rejected instead of resolving by argument order.
|
||||
The plan command uses the same selection contract and prints only the selected
|
||||
range. Planning is read-only: it clones the loaded manifest, models selected
|
||||
stage transitions and invalidation in memory, and invokes resume validation
|
||||
without writing the manifest, creating run directories, materializing files,
|
||||
or invoking pipeline adapters. Analyze detail separates explicit targets,
|
||||
prerequisite rebuilds, scheduled execution, and current reuse. This lets a
|
||||
coarsely stale aggregate analyze stage show zero artifact executions when its
|
||||
selected artifact evidence is still semantically current.
|
||||
|
||||
Before a bounded run or plan whose range starts after `prepare`, every excluded
|
||||
prefix stage must already have a session-manifest status of `succeeded` or
|
||||
`skipped`. Narratio reports the first absent, pending, running, failed, stale,
|
||||
or interrupted prerequisite without creating a run record or changing session
|
||||
state. Widen `--from` to include that stage, or recover it explicitly before
|
||||
retrying. Excluded prefix stages are not resume-validated or repaired as part
|
||||
of the bounded invocation; selected stages still reject missing, unsafe, or
|
||||
manifest-inconsistent inputs at their owning boundary.
|
||||
|
||||
Stages after `--through` are not prerequisites and are never scheduled by the
|
||||
bounded invocation. A selected forced stage can mark one of those succeeded
|
||||
dependents stale through the fixed invalidation relation, but the dependent
|
||||
does not execute until a later invocation selects it. Production composition
|
||||
likewise initializes only collaborators needed by the selected range and
|
||||
shared session lifecycle. In particular, render does not require Notarius or
|
||||
Scriptorium, extract does not require Scriptorium, and analyze does not require
|
||||
the transcription, Seriatim, Audita, or Notarius adapters.
|
||||
|
||||
For the common post-transcript development loop, use:
|
||||
|
||||
```bash
|
||||
narratio regenerate-artifacts 2026-04-04
|
||||
narratio regenerate-artifacts 2026-04-04 --artifacts session_recap,player_handout
|
||||
```
|
||||
|
||||
This command is a transparent expansion to a forced bounded `run` from
|
||||
`extract` through `analyze`. Extraction always rebuilds its complete configured
|
||||
bundle. Analysis rebuilds the selected targets and their required analysis
|
||||
prerequisites, or uses the normal default selection when no artifact names are
|
||||
given. The command does not run publish or notify; delivery remains a separate
|
||||
operator action.
|
||||
|
||||
Inspect current artifact evidence, then publish explicitly when the regenerated
|
||||
set is ready:
|
||||
|
||||
```bash
|
||||
narratio session artifacts 2026-04-04
|
||||
narratio publish 2026-04-04
|
||||
```
|
||||
|
||||
If planning or execution reports stale, missing, failed, legacy, or tampered
|
||||
analysis evidence, regenerate the affected target instead of copying an older
|
||||
canonical file into place or editing the manifest. See
|
||||
[Troubleshooting: Analysis artifact evidence is not current](./troubleshooting.md#analysis-artifact-evidence-is-not-current).
|
||||
|
||||
## Artifact Selection
|
||||
|
||||
`--artifacts` can be used on `run`, `run-stage`, `analyze`, and `publish`.
|
||||
`--artifacts` can be used on `run`, `session plan`, `run-stage`, `analyze`, and
|
||||
`publish`. For a bounded run or plan, the selected range must contain `analyze`
|
||||
or `publish`.
|
||||
|
||||
Selection behavior:
|
||||
|
||||
- validates names against `pipeline.scriptorium.artifacts`;
|
||||
- filters analyze execution to selected configured artifacts;
|
||||
- selects explicit analyze targets and permits their required configured
|
||||
prerequisites to be reused or rebuilt first;
|
||||
- filters publish rules for `narratio.artifact.<name>` sources only;
|
||||
- does not suppress built-in transcript, bounds, or explicitly configured
|
||||
`narratio.extraction.<name>` publish sources; and
|
||||
@@ -127,29 +207,70 @@ The directory is immutable once promoted. Configured lanes become
|
||||
the bundle and `index.json` are retained for audit and resume validation but
|
||||
are not selectable or published implicitly.
|
||||
|
||||
Configured Notarius references resolve only from the current manifest-backed
|
||||
prepared inputs. Their canonical locations are `inputs/party.yml`,
|
||||
`inputs/players.yml`, `inputs/glossary.yml`, and, when configured,
|
||||
`inputs/spell_catalog.json`. Extraction supplies Notarius with verified copies
|
||||
under `runs/<run_id>/extract/references/` so a concurrent refresh of canonical
|
||||
prepared files cannot change the bytes consumed by an in-flight invocation.
|
||||
Inspect the effective stable-input inventory and
|
||||
prepared-file readiness with:
|
||||
|
||||
```bash
|
||||
narratio session status 2026-04-04
|
||||
narratio session validate 2026-04-04
|
||||
```
|
||||
|
||||
Reference metadata records selector, source ID, session-relative path,
|
||||
checksum, and byte size, but never payload contents. Changing a prepared
|
||||
reference changes extraction identity: ordinary continuation rejects the old
|
||||
result, reruns Notarius, and marks successful downstream stages stale. If the
|
||||
prepared file is missing or inconsistent with its manifest checksum, repair
|
||||
the source configuration and refresh prepared state first:
|
||||
|
||||
```bash
|
||||
narratio run-stage prepare 2026-04-04 --force
|
||||
```
|
||||
|
||||
Starting a replacement clears the previous extraction payload from the current
|
||||
session-stage record. If that replacement fails or self-skips, the current
|
||||
record does not fall back to the earlier outputs. The earlier run manifest and
|
||||
immutable bundle remain available for inspection, but downstream resolution
|
||||
requires a new current successful extraction record.
|
||||
|
||||
Atomic Notarius bundle promotion is supported on Linux, macOS, and Windows.
|
||||
On other operating systems, extraction fails before copying the bundle into a
|
||||
Atomic Notarius bundle promotion is supported on Linux and macOS. On Windows
|
||||
and other operating systems, extraction fails before copying the bundle into a
|
||||
temporary promotion tree because Narratio has no verified atomic no-replace
|
||||
directory primitive there. This is an extraction limitation, not a broader
|
||||
platform-support guarantee for every Narratio workflow.
|
||||
|
||||
## External Command Lifecycle
|
||||
|
||||
When an external command is cancelled or times out, Narratio terminates its
|
||||
owned descendants as well as the command itself. Cancellation first requests
|
||||
termination where the platform supports it, then force terminates after a
|
||||
bounded wait. A command is not considered finished until its leader has been
|
||||
reaped, and descendants that keep standard output or error open cannot keep
|
||||
the invocation blocked. Other operating systems fail closed rather than launch
|
||||
a command without tree ownership.
|
||||
|
||||
Subprocess stdout and stderr diagnostics are separately redacted and capped at
|
||||
8 MiB per invocation. Narratio does not retain configured credential values in
|
||||
these logs or their error tails; reaching a capture limit terminates the command
|
||||
tree and reports which stream exceeded the limit.
|
||||
|
||||
Run-local diagnostics are:
|
||||
|
||||
- `runs/{run_id}/extract/notarius.receipt.json`
|
||||
- `runs/{run_id}/extract/notarius.stderr.log`
|
||||
- `runs/{run_id}/extract/notarius-output/` before durable promotion
|
||||
|
||||
The run-record upload excludes the complete
|
||||
`extract/notarius-output/**` subtree. The receipt and stderr files remain
|
||||
eligible run-record diagnostics. The durable bundle is never scanned for
|
||||
implicit publication; only lanes named by explicit `pipeline.publish.outputs`
|
||||
rules are uploaded.
|
||||
The run-record upload is an allowlist derived from the validated run manifest,
|
||||
not a workspace scan. Each declared source is opened without following
|
||||
symlinked ancestors or the leaf, verified as a regular file, and streamed from
|
||||
that verified descriptor. Unlisted files and unsafe entries are never uploaded.
|
||||
The durable bundle is never scanned for implicit publication; only lanes named
|
||||
by explicit `pipeline.publish.outputs` rules are uploaded.
|
||||
|
||||
To intentionally replace the current extraction result, run:
|
||||
|
||||
@@ -157,11 +278,12 @@ To intentionally replace the current extraction result, run:
|
||||
narratio run-stage extract 2026-04-04 --force
|
||||
```
|
||||
|
||||
Narratio automatically reruns extraction when its recorded invocation contract
|
||||
or durable output validation changes. It cannot fingerprint configuration
|
||||
files, profiles, prompts, modules, or references loaded transitively by
|
||||
Notarius. Force extraction after changing any of those inputs, even when the
|
||||
top-level Narratio and Notarius config paths remain the same. A forced extract
|
||||
Narratio automatically reruns extraction when its recorded invocation contract,
|
||||
prepared Narratio reference identities, or durable output validation changes.
|
||||
It cannot fingerprint configuration files, profiles, prompts, modules, or
|
||||
other references loaded transitively by Notarius itself. Force extraction after
|
||||
changing any of those inputs, even when the top-level Narratio and Notarius
|
||||
config paths remain the same. A forced extract
|
||||
marks successful downstream stages stale. Ordinary extraction failures or
|
||||
outcome changes also stale affected downstream stages, while an identical
|
||||
repeated `notarius_disabled` self-skip does not repeatedly invalidate them.
|
||||
@@ -184,13 +306,29 @@ Publish commit model:
|
||||
|
||||
- uploads eligible run files under `{session_prefix}/runs/{run_id}/`, excluding
|
||||
audio and the run-local Notarius staging bundle;
|
||||
- uploads configured published outputs, including only explicitly configured
|
||||
extraction lanes;
|
||||
- uploads `previous/**` cache files when present;
|
||||
- writes `current/manifest.json`;
|
||||
- writes `current/run_id.txt` last.
|
||||
- uploads configured published outputs and `previous/**` cache files into the
|
||||
same immutable run scope, including only explicitly configured extraction
|
||||
lanes;
|
||||
- writes `{session_prefix}/runs/{run_id}/commit.json` after all declared
|
||||
immutable objects are uploaded and verified; and
|
||||
- writes `{session_prefix}/current/commit-pointer.json` once, last.
|
||||
|
||||
`current/run_id.txt` is the remote current-state commit marker.
|
||||
`current/commit-pointer.json` is the remote current-state commit marker. It
|
||||
selects exactly one immutable commit, which declares the complete object set.
|
||||
|
||||
## Remote Commit Migration
|
||||
|
||||
The immutable remote commit contract uses
|
||||
`runs/{run_id}/commit.json` to declare a run's complete object set and a small
|
||||
`current/commit-pointer.json` to select it. The pointer binds the selected
|
||||
commit by version, checksum, size, and storage generation; committed artifacts
|
||||
are also checksum- and generation-bound. Readers accept this contract now and
|
||||
strictly reject mismatched or unknown data.
|
||||
|
||||
Legacy reads are limited to a coherent `current/manifest.json` and
|
||||
`current/run_id.txt` pair; a torn pair is rejected. New publication does not
|
||||
write that pair and remote commit state does not carry local
|
||||
`current_pointer_written` metadata.
|
||||
|
||||
## Publish Locks
|
||||
|
||||
@@ -204,7 +342,14 @@ Effective lock rules:
|
||||
- static and remote locks are merged;
|
||||
- static locks win on source collisions;
|
||||
- locked outputs are intentional skips;
|
||||
- lock add/remove commands mutate only remote lock state.
|
||||
- lock add/remove commands mutate only remote lock state through generation-bound
|
||||
conditional writes. A command retries a bounded number of concurrent
|
||||
conflicts while its invocation context remains active, so it never replaces a
|
||||
different lock-document generation; and
|
||||
- a publish re-reads remote locks immediately before it writes the current
|
||||
commit pointer. A lock committed before that recheck prevents selecting the
|
||||
new snapshot, even though its already-uploaded immutable objects may remain
|
||||
available for a later retry.
|
||||
|
||||
Examples:
|
||||
|
||||
@@ -230,20 +375,30 @@ Apply:
|
||||
narratio session restore 2026-04-04
|
||||
```
|
||||
|
||||
`--dry-run` does not write durable session files. It still reads the selected
|
||||
remote current state and may read object identity/content needed to classify the
|
||||
plan, so it is not a network-free operation.
|
||||
|
||||
Default restore scope:
|
||||
|
||||
- `manifest.json`
|
||||
- `transcripts/**`
|
||||
- `artifacts/**`
|
||||
- the committed session manifest and the committed transcript/artifact objects
|
||||
declared by the selected remote commit
|
||||
- `previous/**` when needed by configured previous-session artifact inputs
|
||||
|
||||
Optional:
|
||||
|
||||
- `--include-audio` to include `audio/**`
|
||||
- `--force` to overwrite local conflicts
|
||||
- `--force` to overwrite eligible conflicting regular files; it never replaces
|
||||
directories or other non-regular local targets
|
||||
|
||||
Restore writes an execution report at `reports/restore-latest.json`.
|
||||
|
||||
If restore fails after beginning installation, it leaves a durable
|
||||
`.restore-incomplete.json` marker in the session root. Pipeline runs will stop
|
||||
until you rerun the same restore command and it completes. Restore intentionally
|
||||
does not try to roll back files already installed; retrying the selected remote
|
||||
snapshot is the recovery procedure.
|
||||
|
||||
## Local State Layout
|
||||
|
||||
Session root:
|
||||
@@ -284,6 +439,38 @@ Cache layout (durable S3 audio cache):
|
||||
|
||||
- `{cache.root}/s3/{bucket}/...`
|
||||
|
||||
Each cached audio file has an adjacent managed identity record. It binds the
|
||||
file to its remote object version and verified digest; deleting or altering the
|
||||
record simply causes Narratio to download and verify the object again.
|
||||
|
||||
### Workspace Permissions
|
||||
|
||||
Ordinary Narratio workspace content is intentionally shareable with the
|
||||
workspace group. On POSIX systems, Narratio-created workspace, spool, and cache
|
||||
directories converge on setgid `02775`; ordinary files, including manifests,
|
||||
transcripts, generated configuration, logs, reports, and Notarius artifacts,
|
||||
converge on `0664`. Narratio explicitly applies these modes so a restrictive
|
||||
caller umask does not remove group write or setgid. It does not change file or
|
||||
directory ownership: the configured workspace's existing group is inherited.
|
||||
|
||||
Windows does not implement POSIX mode bits or setgid semantics. Configure the
|
||||
workspace, spool, and cache locations with an ACL that grants the collaborating
|
||||
group read/write access, and configure credential locations with an ACL limited
|
||||
to the intended credential owner. Do not use POSIX mode displays as evidence of
|
||||
Windows access control.
|
||||
|
||||
API keys are credentials, not ordinary workspace data. Store them outside the
|
||||
shared workspace or in a separately restricted credential location; ordinary
|
||||
workspace group access must never be treated as authorization to read keys.
|
||||
On POSIX, provision a credential directory as `0700` and credential files as
|
||||
`0600`; Narratio rejects group- or other-readable configured credential paths.
|
||||
On Windows, restrict the directory and files with ACLs to the credential owner.
|
||||
|
||||
External adapter results are individually bounded before Narratio validates or
|
||||
materializes them. These per-file limits do not reserve disk space: prevent hard
|
||||
disk exhaustion with filesystem, service, container, or volume quotas sized for
|
||||
the session workload.
|
||||
|
||||
## Cleanup
|
||||
|
||||
Session-scoped cleanup:
|
||||
@@ -309,9 +496,18 @@ Rules:
|
||||
|
||||
- `clean` deletes work/spool session state;
|
||||
- cache is preserved unless `--clear-cache` is set;
|
||||
- each deletion is confined beneath its configured workspace, spool, or cache
|
||||
root and refuses symlinked paths;
|
||||
- automatic post-publish cleanup is gated by successful publish commit plus:
|
||||
- `pipeline.spool.delete_audio_after_publish=true`
|
||||
- `pipeline.workspace.cleanup_after_publish=true`
|
||||
- Narratio first records the exact run-scoped cleanup obligation. If cleanup
|
||||
reports incomplete, the remote committed snapshot remains current; rerun
|
||||
publish to retry only the outstanding confined local cleanup.
|
||||
|
||||
Post-publish cleanup is evaluated only when `publish` actually executes in the
|
||||
current invocation. A bounded range that excludes publish does not replay a
|
||||
cleanup obligation as an unrelated side effect.
|
||||
|
||||
## Operational Caveats
|
||||
|
||||
|
||||
@@ -29,6 +29,10 @@ in the [integration documentation](../integrations/).
|
||||
The pipeline has one canonical ordered stage set. Configuration may enable,
|
||||
disable, or parameterize supported behavior, but it must not turn that sequence
|
||||
into an arbitrary DAG or hide orchestration in generic workflow abstractions.
|
||||
An invocation selects either the full sequence or one inclusive contiguous
|
||||
range of it. Execution remains flat and canonical even though invalidation is
|
||||
dependency-aware: the application owns a separate fixed relation used only to
|
||||
stale transitive dependents, including dependents outside a selected range.
|
||||
The implemented stage inventory belongs in the
|
||||
[Internal Overview](../internal/overview.md).
|
||||
|
||||
@@ -87,8 +91,10 @@ merely on incidental files existing on disk.
|
||||
|
||||
A failed or interrupted stage must not be presented as successful. Failure
|
||||
should preserve enough local state and diagnostics for inspection, recovery,
|
||||
and resume. Forcing an upstream stage invalidates succeeded downstream work
|
||||
according to the canonical stage order.
|
||||
and resume. Forcing a stage invalidates succeeded transitive dependents
|
||||
according to a fixed application-owned relation that is separate from canonical
|
||||
execution order. The relation is validated against the stage inventory and is
|
||||
not configurable.
|
||||
|
||||
A stage may explicitly self-skip with a stable reason and no outputs. That
|
||||
outcome is persisted, clears older outputs owned by the stage, and is
|
||||
@@ -119,6 +125,17 @@ install the validated session manifest after other restored durable files. The
|
||||
physical workflow and recovery procedures belong in
|
||||
[Operations](../operations.md).
|
||||
|
||||
Restore and runner transitions for one session use the same local lock. A
|
||||
durable incomplete-restore marker blocks runner reuse after a partial restore;
|
||||
safe retry, rather than rollback of arbitrary local effects, is the recovery
|
||||
mechanism. Restored manifest-local references must be confined to the selected
|
||||
local session root, never trusted as producer-machine absolute paths.
|
||||
|
||||
For the immutable remote-commit protocol, a restore or status operation binds
|
||||
to one pointer-selected commit and only its declared object identities. A force
|
||||
flag may replace an eligible regular managed file, but never turns a directory
|
||||
or other non-regular conflict into a successful restore.
|
||||
|
||||
## Configuration
|
||||
|
||||
Configuration is strict, explicit, centralized, and operator-oriented.
|
||||
@@ -146,6 +163,10 @@ Canonical helpers own workspace, spool, cache, session, run, input, transcript,
|
||||
artifact, log, report, configuration, and publish-current paths. Callers must
|
||||
not reconstruct canonical paths through scattered string concatenation.
|
||||
|
||||
Reusable audio cache entries require a typed record that binds a confined,
|
||||
no-follow regular file and its digest to the selected remote object identity.
|
||||
Size alone and unqualified multipart ETags are not content-integrity evidence.
|
||||
|
||||
Artifact resolution is deterministic and manifest-aware. Producers materialize
|
||||
canonical outputs before reporting success, and consumers resolve declared
|
||||
artifact identities rather than infer files from unrelated directory contents.
|
||||
@@ -167,22 +188,38 @@ contracts belong under [Integrations](../integrations/).
|
||||
## Publish Commit Boundary
|
||||
|
||||
Publish has one explicit remote commit boundary. A remote run becomes current
|
||||
only after Narratio has successfully uploaded the run record, required published
|
||||
outputs, `current/manifest.json`, and finally `current/run_id.txt`.
|
||||
only after Narratio has successfully uploaded its immutable run-scoped objects,
|
||||
the immutable commit manifest, and finally the current commit pointer.
|
||||
|
||||
`current/run_id.txt` is the commit marker and must be written last. Failed,
|
||||
incomplete, skipped, or uncommitted publish attempts must not be presented as
|
||||
current remote state. Publish locks remain authoritative and are not bypassed by
|
||||
a forced run.
|
||||
`current/commit-pointer.json` is the sole mutable selector and must be written
|
||||
exactly once, last. Failed, incomplete, skipped, or uncommitted publish attempts
|
||||
must not be presented as current remote state. Publish locks remain authoritative
|
||||
and are not bypassed by a forced run. Mutable remote locks use provider-enforced
|
||||
generation preconditions and are revalidated immediately before pointer
|
||||
selection; loss of that check leaves the prior committed snapshot current.
|
||||
|
||||
Automatic local cleanup is permitted only after a successful publish commit,
|
||||
only when explicitly configured, and only through the path-safety guardrails.
|
||||
It is a durable local obligation bound to that committed run and its exact
|
||||
targets, not an inferred side effect of the current stage list. A cleanup
|
||||
failure makes the invocation incomplete while leaving the committed remote
|
||||
snapshot authoritative; later invocations resume the recorded obligation.
|
||||
|
||||
## Security, Privacy, And Diagnostics
|
||||
|
||||
Narratio handles private campaign material. Transcripts, prompts, generated
|
||||
artifacts, reports, logs, manifests, and diagnostic files are potentially
|
||||
sensitive.
|
||||
Narratio distinguishes ordinary workspace data from credentials. Campaign and
|
||||
session material—including manifests, transcripts, prompts, generated
|
||||
configuration, logs, reports, diagnostics, and Notarius artifacts—is
|
||||
intentionally shareable with the configured workspace group. API-key material
|
||||
is sensitive and is not covered by the ordinary workspace-sharing policy.
|
||||
|
||||
On POSIX systems, Narratio-created ordinary workspace directories converge on
|
||||
setgid `02775` and ordinary workspace files on `0664`, even when the caller's
|
||||
umask is restrictive. This preserves the existing workspace group for nested
|
||||
creation and atomic replacements without changing ownership. API-key storage
|
||||
uses a separate restrictive contract. On Windows, POSIX mode bits and setgid
|
||||
are not authoritative; operators must provide the equivalent shared-group and
|
||||
credential-restricted ACLs described in [Operations](../operations.md#workspace-permissions).
|
||||
|
||||
Raw secrets must not be stored in pipeline, campaign, or session YAML or written
|
||||
to manifests, logs, generated configuration, reports, publish metadata,
|
||||
|
||||
7
docs/releases/README.md
Normal file
7
docs/releases/README.md
Normal file
@@ -0,0 +1,7 @@
|
||||
# Release Notes
|
||||
|
||||
This directory contains the maintained release-note text for Narratio releases.
|
||||
The corresponding Gitea release is the canonical source for downloadable
|
||||
binaries and checksums.
|
||||
|
||||
- [v1.5.0](v1.5.0.md)
|
||||
47
docs/releases/v1.5.0.md
Normal file
47
docs/releases/v1.5.0.md
Normal file
@@ -0,0 +1,47 @@
|
||||
# Narratio v1.5.0
|
||||
|
||||
Narratio v1.5.0 makes repeated post-transcript artifact development faster and
|
||||
more explicit while retaining the fixed, stage-driven pipeline model.
|
||||
|
||||
## Highlights
|
||||
|
||||
- The canonical pipeline now completes deterministic rendering before
|
||||
extraction, cleanly separating transcript-generating stages from
|
||||
artifact-generating stages.
|
||||
- `narratio run` and `narratio session plan` accept inclusive `--from` and
|
||||
`--through` bounds. Excluded transcript stages are not executed or
|
||||
invalidated by a bounded artifact-regeneration run.
|
||||
- `narratio regenerate-artifacts SESSION` is an exact convenience alias for a
|
||||
forced run from `extract` through `analyze`, including focused
|
||||
`--artifacts` selections.
|
||||
- Configured Scriptorium artifacts now have independent,
|
||||
manifest-authoritative freshness. Narratio reuses validated current work,
|
||||
rebuilds stale prerequisites in dependency order, and persists successful,
|
||||
failed, and newly stale artifact state when an analysis invocation only
|
||||
partially succeeds.
|
||||
- Publish consumes only configured artifacts backed by current manifest
|
||||
evidence; incidental or tampered files are not promoted as current output.
|
||||
|
||||
## Reliability And Administration
|
||||
|
||||
- Bounded prerequisites are checked again under the session lock before any
|
||||
run mutation, closing a concurrent-run race.
|
||||
- Analysis fingerprints are stable across executable and configuration path
|
||||
changes and continue to cover only Narratio-observable semantic inputs.
|
||||
- Runner composition now carries one validated execution plan from command
|
||||
parsing through prerequisite validation, adapter composition, manifest
|
||||
recording, and stage execution.
|
||||
- `narratio version` reports the exact tag embedded in official release
|
||||
binaries; ordinary source builds report `dev`.
|
||||
|
||||
## Upgrade Notes
|
||||
|
||||
- Existing unbounded commands and direct `run-stage`, `analyze`, and `publish`
|
||||
workflows retain their meanings.
|
||||
- Manifests written before artifact-level analysis state remain readable.
|
||||
Legacy aggregate analysis success is not sufficient freshness evidence, so
|
||||
the first analysis evaluation after upgrading may regenerate configured
|
||||
artifacts once.
|
||||
- Narratio cannot observe executable contents or configuration, prompt,
|
||||
profile, module, and other files loaded privately by Scriptorium. Explicitly
|
||||
force affected artifacts after changing those private inputs.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,332 +0,0 @@
|
||||
# Codebase Audit Plan
|
||||
|
||||
Status: proposed
|
||||
|
||||
## Purpose
|
||||
|
||||
This audit will evaluate Narratio for correctness, efficiency, maintainability,
|
||||
and test-suite value. It will identify defects and credible risks, duplicated or
|
||||
near-duplicated behavior, code that can be made smaller or more idiomatic, and
|
||||
complex code whose remaining invariants need focused explanation.
|
||||
|
||||
The audit is investigative. It should produce evidence-backed findings and a
|
||||
prioritized remediation backlog, not make opportunistic production changes as
|
||||
it proceeds. The [Audit Sequence](audit-sequence.md) assigns this scope to
|
||||
concrete execution stages.
|
||||
|
||||
## Authoritative Baseline
|
||||
|
||||
Review implemented behavior against its canonical owner rather than treating
|
||||
the current implementation or tests as the specification:
|
||||
|
||||
- [Architecture](../policy/architecture.md) for system boundaries, dependency
|
||||
direction, state and path ownership, safety properties, and pipeline
|
||||
invariants;
|
||||
- [Internal Overview](../internal/overview.md) and its focused internal
|
||||
documents for implemented ownership and mechanics;
|
||||
- [Testing Policy](../policy/testing.md) for risk-based sufficiency, durable
|
||||
boundaries, test-double guidance, and test lifecycle decisions;
|
||||
- the [CLI](../cli.md), [Configuration](../config.md),
|
||||
[Operations](../operations.md), and [integration contracts](../integrations/)
|
||||
for externally observable behavior; and
|
||||
- the [Documentation Policy](../policy/documentation.md) for canonical ownership
|
||||
and the distinction between current and proposed behavior.
|
||||
|
||||
Where code, tests, and documentation disagree, record the disagreement. Do not
|
||||
assume which one is wrong until the canonical contract and caller expectations
|
||||
have been traced.
|
||||
|
||||
## Audit Principles
|
||||
|
||||
1. Review correctness before cleanup. A shorter implementation is not an
|
||||
improvement if it weakens a state transition, safety check, or external
|
||||
contract.
|
||||
2. Trace behavior across boundaries. Narratio's most important properties often
|
||||
emerge from the interaction of application orchestration, stages, manifests,
|
||||
artifact resolution, filesystem operations, and adapters.
|
||||
3. Distinguish repeated syntax from repeated policy. Extract a helper only when
|
||||
the behavior has one stable owner and the shared abstraction makes that
|
||||
ownership clearer. Similar stage code may be intentionally explicit.
|
||||
4. Prefer narrow, idiomatic Go over generic frameworks. In particular, proposed
|
||||
refactors must preserve the explicit canonical stage sequence and must not
|
||||
turn Narratio into a workflow engine or a second configuration system for
|
||||
downstream tools.
|
||||
5. Optimize credible work. Flag repeated I/O, hashing, serialization, remote
|
||||
calls, subprocess work, allocation, or poor asymptotic behavior when the
|
||||
relevant path can matter. Require a benchmark or workload argument for
|
||||
performance changes whose benefit is not evident.
|
||||
6. Treat comments as explanations of intent. Recommend comments for invariants,
|
||||
ordering constraints, non-obvious failure policy, or security reasoning—not
|
||||
as narration of ordinary Go or a substitute for simplifying code.
|
||||
7. Judge tests as a suite. A test can be locally reasonable and still add no
|
||||
marginal protection, while a compact test can be inadequate for a
|
||||
consequential cross-component failure.
|
||||
|
||||
## Evidence And Finding Standard
|
||||
|
||||
Begin from a cleanly identified revision and record toolchain and platform
|
||||
assumptions. Use the code knowledge graph to find ownership, callers, callees,
|
||||
similarity candidates, high-complexity functions, and weakly protected
|
||||
boundaries. Confirm every candidate by reading the implementation, its focused
|
||||
tests, and the applicable contract. Text search and static analysis supplement
|
||||
the graph for literals, configuration, generated files, and patterns that are
|
||||
not modeled reliably.
|
||||
|
||||
Each finding should record:
|
||||
|
||||
- category: correctness defect, correctness risk, duplication, simplification,
|
||||
efficiency, architectural boundary, comment/clarity, or test-suite issue;
|
||||
- source locations and the affected contract or invariant;
|
||||
- concrete evidence and a realistic failure or maintenance scenario;
|
||||
- impact, likelihood, confidence, and estimated remediation scope separately;
|
||||
- the smallest plausible improvement and its intended owner;
|
||||
- tests that already protect the behavior, tests that should change or be
|
||||
added, and tests that may become redundant; and
|
||||
- dependencies on, or conflicts with, other findings.
|
||||
|
||||
Do not report a metric alone as a finding. Complexity, similarity, coverage,
|
||||
fan-in, file size, and test count are prioritization signals that require manual
|
||||
confirmation. Consolidate findings that share one root cause.
|
||||
|
||||
## Cross-Cutting Review Lenses
|
||||
|
||||
### Correctness And Pipeline Semantics
|
||||
|
||||
Construct an explicit lifecycle matrix for every stage outcome: first run,
|
||||
already-succeeded skip, self-skip, failure, interruption, forced replacement,
|
||||
non-resumable result, and successful rerun. Trace how each outcome changes the
|
||||
session manifest, invocation manifest, downstream stage state, artifacts,
|
||||
diagnostics, and cleanup eligibility.
|
||||
|
||||
Across the pipeline, verify:
|
||||
|
||||
- the registry exposes one deterministic canonical order;
|
||||
- each stage's declared inputs, outputs, configuration, adapters, and manifest
|
||||
effects agree with its implementation and focused documentation;
|
||||
- inputs are resolved through manifest and artifact contracts rather than
|
||||
incidental directory contents;
|
||||
- run-local outputs are fully validated before canonical materialization;
|
||||
- failure, cancellation, or process interruption cannot advertise partial work
|
||||
as successful;
|
||||
- force and changed outcomes invalidate exactly the intended succeeded
|
||||
downstream work;
|
||||
- repeated execution is idempotent where promised, and ordering is stable
|
||||
wherever maps, directory reads, remote listings, or dependency graphs are
|
||||
involved;
|
||||
- session, campaign, run, source, checksum, contract, and external provenance
|
||||
identities cannot be confused across runs; and
|
||||
- errors preserve useful causes and do not expose secrets or private content.
|
||||
|
||||
Use fault-oriented reasoning at durability boundaries: fail immediately before
|
||||
and after manifest saves, canonical renames, external process completion,
|
||||
uploads, current-manifest publication, the current-run commit marker, restore
|
||||
manifest installation, and cleanup. Determine which state is authoritative and
|
||||
whether the next invocation recovers safely.
|
||||
|
||||
### Duplication And Helper Ownership
|
||||
|
||||
Search for exact and semantic duplication in production and tests, including:
|
||||
|
||||
- repeated stage setup, input resolution, output validation, run-local
|
||||
materialization, metadata construction, and error adaptation;
|
||||
- repeated manifest create/load/save and session/run transition handling;
|
||||
- repeated adapter construction, timeout parsing, command execution, generated
|
||||
configuration, log handling, and output checks;
|
||||
- repeated source-ID, destination, remote-key, and path validation policy;
|
||||
- repeated sorting, deduplication, checksum, copy, and atomic-write mechanics;
|
||||
and
|
||||
- repeated test fixtures and assertions that encode the same policy at several
|
||||
layers.
|
||||
|
||||
For each candidate, decide whether it is coincidental similarity, a repeated
|
||||
mechanism, or duplicated policy. Recommend extraction only when the helper can
|
||||
have a clear package owner, a narrow contract, and callers that become easier
|
||||
to understand. Prefer an unexported local helper when sharing is package-local.
|
||||
Do not create a broad utility package, force unlike stage results into one data
|
||||
model, or move policy into storage/file-operation helpers.
|
||||
|
||||
Initial similarity and complexity signals should seed, but not predetermine,
|
||||
inspection of the single-stage command wrappers, session/run manifest
|
||||
persistence pairs, adapter constructors, Scriptorium operations, stage fakes,
|
||||
and common stage materialization paths.
|
||||
|
||||
### Simplification, Go Idioms, And Efficiency
|
||||
|
||||
Review long or branch-heavy functions for separable decisions, state
|
||||
transitions, or data transformations. Pay particular attention to orchestration,
|
||||
configuration validation, artifact dependency resolution, resume verification,
|
||||
restore/previous-cache planning, and analyze/publish selection logic. A useful
|
||||
refactor should reduce cognitive load while leaving the important ordering
|
||||
visible.
|
||||
|
||||
Check for:
|
||||
|
||||
- unnecessary nesting, defensive branches made unreachable by earlier
|
||||
validation, repeated normalization, and overly wide parameter lists;
|
||||
- interfaces defined for hypothetical extensibility rather than a demonstrated
|
||||
consumer boundary;
|
||||
- manual slice, map, string, error, and filesystem logic with a clearer standard
|
||||
library form;
|
||||
- incorrect or inconsistent `errors.Is`/`errors.As`, wrapping, context
|
||||
propagation, deferred cleanup, response-body closure, process waiting, and
|
||||
goroutine/channel ownership;
|
||||
- redundant filesystem scans, `stat`/checksum passes, whole-file buffering,
|
||||
copying, YAML/JSON round trips, sorting, remote listings, downloads, uploads,
|
||||
or adapter initialization;
|
||||
- linear searches nested in loops and repeated dependency or artifact lookup
|
||||
that should use an indexed map or a single planning pass;
|
||||
- unbounded concurrency, leaked work after cancellation, serialized independent
|
||||
work, and nondeterministic result collection; and
|
||||
- obsolete dependencies, portability assumptions, and platform-sensitive path
|
||||
or atomic-rename behavior.
|
||||
|
||||
Keep correctness and diagnosability ahead of micro-optimization. When a simpler
|
||||
algorithm changes performance characteristics, specify the representative
|
||||
input size and validation method.
|
||||
|
||||
### Comments And Local Explanation
|
||||
|
||||
Review high fan-in, high-complexity, security-sensitive, and commit-boundary
|
||||
code after likely simplifications have been identified. Add a comment
|
||||
recommendation when a maintainer needs to know why:
|
||||
|
||||
- state transitions or persistence operations occur in a specific order;
|
||||
- a stale record intentionally retains data while another transition clears it;
|
||||
- a path is checked more than once to resist traversal, symlink replacement, or
|
||||
time-of-check/time-of-use hazards;
|
||||
- an artifact is accepted only with particular manifest, checksum, contract, or
|
||||
provenance evidence;
|
||||
- a partial operation is intentionally not rolled back;
|
||||
- a remote pointer or local manifest must be installed last; or
|
||||
- concurrency, cancellation, compatibility, or downstream-tool behavior makes
|
||||
an apparently simpler approach unsafe.
|
||||
|
||||
Prefer a named helper, typed state, or smaller control flow when that removes the
|
||||
need for explanation. Check existing comments for stale claims as well as
|
||||
missing rationale.
|
||||
|
||||
### Test Suite Against The Canonical Policy
|
||||
|
||||
Build a risk-to-test matrix rather than auditing tests file by file in
|
||||
isolation. For each important behavior, identify its proper owner—parser,
|
||||
validator, domain package, adapter, orchestrator, CLI, integration, or end to
|
||||
end—and identify all tests that claim to protect it.
|
||||
|
||||
Evaluate:
|
||||
|
||||
- protection of data integrity, destructive operations, compatibility,
|
||||
security, concurrency, idempotency, recovery, and partial failure;
|
||||
- manifest transitions, force/invalidation, resume validation, atomic
|
||||
materialization, publish commit order, restore install order, and cleanup
|
||||
gates as assembled behaviors;
|
||||
- realistic HTTP, subprocess, filesystem, and object-store boundary behavior,
|
||||
including cancellation and malformed responses;
|
||||
- whether higher-level tests intentionally sample lower-level behavior or
|
||||
redundantly reproduce its full policy;
|
||||
- whether tests assert durable outcomes or private constants, exact error text,
|
||||
incidental paths, call choreography, or oversized snapshots;
|
||||
- whether real fast collaborators could replace elaborate doubles, and whether
|
||||
stateful fakes are realistic enough for the risk they protect;
|
||||
- fixture/helper duplication, oversized test cases, and setup that obscures the
|
||||
behavior under test without introducing a heavyweight test framework;
|
||||
- deterministic, offline, credential-free, order-independent execution and
|
||||
safe handling of environment and process-global state;
|
||||
- focused fuzz candidates in parsing, normalization, source IDs, remote/local
|
||||
path mapping, manifest decoding, and configuration boundaries; and
|
||||
- the presence and value of a small number of representative assembled
|
||||
workflows.
|
||||
|
||||
Use coverage only to locate unexpectedly weak consequential branches. Also
|
||||
inspect packages with extensive coverage for redundant tests and refactoring
|
||||
friction. For every proposed addition, deletion, or consolidation, state the
|
||||
realistic defect and marginal confidence involved.
|
||||
|
||||
The audit baseline should include the repository's canonical commands plus
|
||||
targeted diagnostic runs where supported:
|
||||
|
||||
```sh
|
||||
go test ./...
|
||||
go test -race ./...
|
||||
go vet ./...
|
||||
go build ./cmd/narratio
|
||||
```
|
||||
|
||||
Use focused repeated or shuffled runs to investigate state leakage and
|
||||
flakiness, and collect package/branch coverage for diagnosis. Review continuous
|
||||
integration to determine whether the appropriate offline validation is enforced;
|
||||
do not turn coverage percentage into a gate merely for this audit.
|
||||
|
||||
## Area-By-Area Inspection Map
|
||||
|
||||
| Area | Primary locations | What to inspect |
|
||||
| --- | --- | --- |
|
||||
| Process and application boundary | `cmd/narratio`, `internal/app` | Command dispatch, configuration selection, production composition, secret loading, lock lifetime, object-store initialization, context/error propagation, and separation of CLI reporting from orchestration policy. Review operator commands for consistent current-state authority and shared read-only mechanics. |
|
||||
| Stage registry and runner | `internal/stage/placeholders.go`, `internal/stage/stage.go`, `internal/app/planner.go`, `internal/app/runner.go`, `internal/app/run_stage.go` | Canonical order, action decisions, resume/force/self-skip/failure transitions, downstream invalidation, session/run manifest consistency, resource lifecycle, cleanup triggering, and opportunities to decompose the runner without hiding its state machine. |
|
||||
| Configuration | `internal/config` | Strict decoding, discovery and precedence, centralized defaults, normalization, templating, validation order, unknown fields, empty-value behavior, secret references, cross-field constraints, path confinement, deterministic errors, duplicated validator policy, and compatibility with maintained examples. |
|
||||
| Prepare and audio | `internal/stage/prepare.go`, `internal/audio`, `internal/previouscache` | Local/S3 exclusivity, cache and spool identity, partial downloads, checksum/reuse policy, previous-session required/optional planning, deterministic input records, clearing semantics, traversal safety, and avoiding repeated remote or filesystem work. |
|
||||
| Transcript stages | `internal/stage/transcribe.go`, `merge.go`, `polish.go`, `normalize.go`, `trim.go`, `render.go` | Contract parity across similar stages, bounded concurrency and cancellation, deterministic speaker/input ordering, run-local validation and canonical promotion, report/diagnostic classification, disabled behavior, and narrow opportunities for shared mechanics. |
|
||||
| Extraction | `internal/stage/extract.go`, `extract_resume.go`, `internal/adapters/notarius`, `internal/fileops/directory.go` | External receipt and lane validation, configuration fingerprint limits, immutable promotion, symlink/root replacement defenses, provenance and checksum checks, immediate and cross-invocation reuse, obsolete versus unsafe outcomes, failure residue, and whether dense verification logic can be clarified without weakening it. |
|
||||
| Analyze and artifact dependencies | `internal/stage/analyze.go`, `internal/artifacts`, `internal/artifactpolicy` | Source-family validation, runtime catalog state, enabled/selected/reused distinctions, topological ordering and cycle handling, required/optional inputs, local-only previous sources, deterministic metadata, repeated lookup/scanning, and ownership shared with config and publish. |
|
||||
| Publish and cleanup | `internal/stage/publish.go`, `internal/app/post_publish_cleanup.go`, `internal/app/cleanup_targets.go` | Prerequisite success, output selection, locks, required/optional behavior, exclusion rules, deterministic upload set, retry/idempotency implications, current-manifest then commit-marker ordering, metadata gates, and destructive path confinement. |
|
||||
| Manifest state | `internal/manifest` | Validation and backward compatibility, atomic persistence, timestamps, session/run identity, transition truth table, clearing versus retaining payload, create/load/save duplication, failure during dual-manifest updates, and whether state mutation has a single owner. |
|
||||
| Artifacts, paths, and policy | `internal/artifacts`, `internal/artifactpolicy`, `internal/pathsafe` | Canonical helper coverage, ad hoc reconstruction by callers, source-ID ownership, manifest-first resolution, extraction/current-state identity, destination normalization, stable ordering, typed missing-state errors, symlink/traversal defenses, and duplicate policy across config/stages/app. |
|
||||
| Restore | `internal/app/restore*.go`, `internal/previouscache`, `internal/audio` | Remote authority, confined mapping, deterministic plan actions, local conflict and force behavior, dry-run purity, temp-file installation, manifest-last ordering, partial failure/retry behavior, report accuracy, cache reuse, and shared current-state mechanics. |
|
||||
| File operations | `internal/fileops`, `internal/pathsafe`, local-store code in `internal/artifacts` | Atomic-write and promotion guarantees, permissions, close/sync/rename error handling, temp cleanup, same-filesystem assumptions, replacement policy, regular-file-only traversal, symlink and root-swap resistance, lock cleanup, and portability. |
|
||||
| External adapters and storage | `internal/adapters`, `internal/audio` | Transport isolation, shared subprocess mechanics versus adapter-specific policy, command/config duplication, quoting and working directories, timeouts/cancellation, stdout/stderr separation, HTTP body and retry behavior, S3 pagination/streaming/not-found mapping, credential independence, and external error adaptation. |
|
||||
| Shared models and diagnostics | `internal/artifactmodel`, `internal/contracts`, `internal/logging` | Serialization and validation invariants, unnecessary conversions, ownership of shared types, stable diagnostic structure, redaction, and whether small shared packages remain cohesive. |
|
||||
| Tests, examples, and automation | all `*_test.go`, `examples/`, `.woodpecker/` | Risk ownership, semantic duplication, fixture cost, policy-coupled assertions, realistic boundary tests, end-to-end sufficiency, default-suite isolation, example validation, diagnostic coverage, flakiness, runtime cost, and enforcement of canonical validation. |
|
||||
|
||||
## Narratio-Specific Cross-Boundary Scenarios
|
||||
|
||||
In addition to package-local review, trace these complete scenarios because a
|
||||
modular pipeline can look correct within every package while violating an
|
||||
end-to-end invariant:
|
||||
|
||||
1. A stage succeeds, its result becomes non-resumable, the rerun fails, and a
|
||||
later invocation decides what remains usable.
|
||||
2. An upstream forced or changed outcome interacts with already-succeeded,
|
||||
self-skipped, and disabled downstream stages.
|
||||
3. Extraction produces a valid immutable bundle, then configuration or
|
||||
transitive Notarius inputs change before analyze or publish.
|
||||
4. Previous-session state is published, restored or prepared into the local
|
||||
cache, and consumed by analyze without an unintended remote read.
|
||||
5. Publish fails at each upload boundary, especially between current manifest
|
||||
and current-run pointer, followed by status, restore, and retry.
|
||||
6. Restore encounters identical files, conflicting files, unsafe remote keys,
|
||||
cache hits, and a failure immediately before manifest installation.
|
||||
7. Automatic or manual cleanup is requested after skipped, failed, locked,
|
||||
partially uploaded, and fully committed publish outcomes.
|
||||
8. Cancellation reaches bounded transcription work, HTTP requests,
|
||||
subprocesses, object storage, and manifest reporting without leaks or false
|
||||
success.
|
||||
9. A configured artifact is disabled, unselected, reused, generated from
|
||||
another artifact, sourced from extraction, or sourced from a previous
|
||||
session, then filtered for publish.
|
||||
10. The same session is invoked concurrently, including lock contention and
|
||||
cleanup/release failures.
|
||||
|
||||
## Completion Criteria
|
||||
|
||||
The audit is complete when:
|
||||
|
||||
- every area in the inspection map has been reviewed against its canonical
|
||||
contracts and focused tests;
|
||||
- the stage lifecycle matrix and cross-boundary scenarios have explicit
|
||||
conclusions;
|
||||
- duplication candidates have been classified rather than merely counted;
|
||||
- simplification and performance recommendations explain their correctness
|
||||
constraints and expected benefit;
|
||||
- comment recommendations identify the non-obvious rationale to preserve;
|
||||
- the test suite has a risk-based sufficiency assessment, including gaps,
|
||||
redundancy, durability, execution properties, and automation;
|
||||
- findings are deduplicated, evidence-backed, and ranked by risk and dependency;
|
||||
and
|
||||
- unresolved questions and intentionally accepted risks are recorded rather
|
||||
than silently omitted.
|
||||
|
||||
## Execution
|
||||
|
||||
The [Audit Sequence](audit-sequence.md) is the canonical owner of execution
|
||||
order, stage boundaries, checkpoints, validation, and audit deliverables. This
|
||||
document remains the canonical owner of audit scope, review criteria, and the
|
||||
finding standard.
|
||||
@@ -1,734 +0,0 @@
|
||||
# Codebase Audit Sequence
|
||||
|
||||
Status: proposed
|
||||
|
||||
## Purpose And Relationship To The Audit Plan
|
||||
|
||||
This document turns the [Codebase Audit Plan](audit-plan.md) into a bounded,
|
||||
execution-ready sequence. The plan owns scope, review criteria, and the finding
|
||||
standard. This document owns ordering, dependencies, working records,
|
||||
validation, and exit gates.
|
||||
|
||||
The sequence is for investigation only. Do not mix production refactors or bug
|
||||
fixes into the audit. A confirmed urgent defect may justify stopping to request
|
||||
a separate remediation change, but its fix is not part of this sequence.
|
||||
|
||||
## Audit Run Records
|
||||
|
||||
Create `docs/roadmap/audit-findings.md` when the audit begins. It is the single
|
||||
working ledger and final audit report. Initialize it with:
|
||||
|
||||
- the audited revision, branch/worktree state, Go version, platform, and audit
|
||||
date;
|
||||
- baseline command results and timings;
|
||||
- an area coverage ledger;
|
||||
- the stage lifecycle matrix;
|
||||
- the cross-boundary scenario matrix from the audit plan;
|
||||
- a risk-to-test matrix;
|
||||
- candidate and confirmed finding registers; and
|
||||
- unresolved questions, accepted risks, and final conclusions.
|
||||
|
||||
Track each execution stage in the coverage ledger with one of `not_started`,
|
||||
`in_progress`, `complete`, or `blocked`. For a completed stage, record:
|
||||
|
||||
- contracts, packages, files, and important symbols reviewed;
|
||||
- graph traces, commands, tests, or other evidence used;
|
||||
- finding and candidate IDs produced;
|
||||
- explicit no-finding conclusions for reviewed high-risk behavior; and
|
||||
- follow-up questions assigned to later stages.
|
||||
|
||||
Use stable finding IDs with these prefixes:
|
||||
|
||||
| Prefix | Category |
|
||||
| --- | --- |
|
||||
| `COR` | Confirmed correctness defect |
|
||||
| `RSK` | Correctness or operational risk |
|
||||
| `ARC` | Ownership or architectural-boundary issue |
|
||||
| `DUP` | Duplicated mechanism or policy |
|
||||
| `SIM` | Simplification or idiomatic-Go opportunity |
|
||||
| `EFF` | Efficiency or resource-use issue |
|
||||
| `COM` | Missing, misleading, or stale explanatory comment |
|
||||
| `TST` | Test-suite gap, redundancy, brittleness, or execution issue |
|
||||
|
||||
Candidate IDs remain candidates until manual inspection confirms the behavior,
|
||||
contract, realistic scenario, and affected callers. Rejected candidates remain
|
||||
in a short classification log so later stages do not reopen them without new
|
||||
evidence.
|
||||
|
||||
## Execution Rules
|
||||
|
||||
1. Pin the audit to the revision recorded in Stage 0. If the worktree or HEAD
|
||||
changes, record the change and rerun every affected stage; do not silently
|
||||
combine evidence from different implementations.
|
||||
2. Use codebase graph search and call/data-flow traces before broad source
|
||||
search. Read the exact implementation, focused tests, and canonical contract
|
||||
before confirming a finding.
|
||||
3. Record test-policy observations during every behavior pass. Stage 12 owns the
|
||||
suite-wide conclusion but must not rediscover the suite from scratch.
|
||||
4. Record cross-area observations as candidates for the stage that owns the
|
||||
conclusion. Avoid producing duplicate findings from several review passes.
|
||||
5. Treat baseline failures as evidence, not automatic blockers. Continue when
|
||||
read-only inspection remains sound, and state the limitation. Stop only when
|
||||
the repository cannot be identified, required sources are unavailable, or a
|
||||
failure makes later evidence unreliable.
|
||||
6. Do not exercise a suspected destructive, credentialed, paid, or live-service
|
||||
path merely to prove a defect. Use source reasoning, existing safe fakes, or
|
||||
a narrowly controlled offline reproduction.
|
||||
7. Escalate a credible active data-loss, secret-exposure, or unsafe-cleanup
|
||||
defect immediately. Preserve the evidence and do not wait for final
|
||||
synthesis before reporting it.
|
||||
8. A stage is complete only when its exit gate is met. A package test passing is
|
||||
evidence, not proof that the review is complete.
|
||||
|
||||
## Sequence Overview
|
||||
|
||||
| Stage | Focus | Depends on | Primary result |
|
||||
| --- | --- | --- | --- |
|
||||
| 0 | Pin revision and establish baseline | None | Reproducible audit record |
|
||||
| 1 | Contract, boundary, and lifecycle map | 0 | Review matrices and ownership map |
|
||||
| 2 | Runner and manifest state machine | 1 | Lifecycle and dual-ledger conclusions |
|
||||
| 3 | Paths, artifacts, and filesystem safety | 1-2 | State/path authority and mutation conclusions |
|
||||
| 4 | Publish, remote commit, and cleanup | 2-3 | Commit-boundary and destructive-operation conclusions |
|
||||
| 5 | Restore and remote/previous state | 2-4 | Restore authority and recovery conclusions |
|
||||
| 6 | Configuration and application composition | 1-5 | Validation and wiring conclusions |
|
||||
| 7 | External adapters and shared support | 3, 6 | Boundary, cancellation, and resource conclusions |
|
||||
| 8 | Prepare and transcript-processing stages | 2-3, 6-7 | Ordinary stage-contract conclusions |
|
||||
| 9 | Extraction vertical slice | 2-3, 6-7 | Promotion, provenance, and resume conclusions |
|
||||
| 10 | Analyze and artifact dependency slice | 3, 6, 8-9 | Dependency and source-resolution conclusions |
|
||||
| 11 | Cross-codebase duplication, simplicity, efficiency, and comments | 2-10 | Classified maintainability candidates |
|
||||
| 12 | Test-suite policy audit | 2-11 | Risk-based suite sufficiency assessment |
|
||||
| 13 | Synthesis and audit closeout | 0-12 | Final deduplicated audit report |
|
||||
|
||||
Stages are intentionally ordered. Later stages may resolve candidates raised by
|
||||
earlier ones, but they must not invalidate an earlier stage silently. Return to
|
||||
the owning stage, update its coverage record, and note the new evidence.
|
||||
|
||||
## Stage 0: Pin Revision And Establish Baseline
|
||||
|
||||
### Entry
|
||||
|
||||
- Repository root and `docs/development.md` are available.
|
||||
- The audit plan and canonical policy documents can be read.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Record `git rev-parse HEAD`, branch/detached state, `git status --short`,
|
||||
`go version`, `go env GOOS GOARCH`, and the current date.
|
||||
2. Confirm that the code knowledge graph represents the recorded repository and
|
||||
revision; refresh the index if it is missing or stale.
|
||||
3. Capture the package/file/test inventory, entry points, architecture
|
||||
boundaries, high fan-in symbols, complexity signals, and similarity signals.
|
||||
4. Run the default offline baseline and record wall time and failures:
|
||||
|
||||
```sh
|
||||
go test -count=1 ./...
|
||||
go test -race -count=1 ./...
|
||||
go vet ./...
|
||||
```
|
||||
|
||||
5. Build into an external temporary directory so validation does not add a
|
||||
workspace binary:
|
||||
|
||||
```sh
|
||||
audit_build_dir="$(mktemp -d)"
|
||||
go build -o "$audit_build_dir/narratio" ./cmd/narratio
|
||||
go test -coverprofile="$audit_build_dir/coverage.out" ./...
|
||||
```
|
||||
|
||||
6. Inventory the repository's CI/release validation, maintained examples, fuzz
|
||||
tests, golden data, opt-in tests, and generated-test update mechanisms.
|
||||
|
||||
### Output
|
||||
|
||||
- Baseline and inventory sections in `audit-findings.md`.
|
||||
- Initial coverage ledger containing Stages 0-13.
|
||||
- Unconfirmed metric-driven candidates, clearly labeled as such.
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Revision and environment are reproducible.
|
||||
- Every baseline command has a recorded result.
|
||||
- Graph freshness is known.
|
||||
- Any limitation that affects later stages has an owner and disposition.
|
||||
|
||||
## Stage 1: Build The Contract, Boundary, And Lifecycle Map
|
||||
|
||||
### Entry
|
||||
|
||||
- Stage 0 is complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Read the architecture, internal overview, testing policy, focused internal
|
||||
documents, and the relevant CLI/configuration/operations/integration
|
||||
contracts using the development guide's routing rules.
|
||||
2. Map each package and important interface to its owned policy. Mark every
|
||||
cross-package dependency that appears to reverse or blur the intended
|
||||
direction for later confirmation.
|
||||
3. Build a stage-contract matrix with canonical order, declared inputs,
|
||||
outputs, configuration, adapters, skip behavior, resume validation,
|
||||
materialization boundary, manifest effects, and downstream invalidation.
|
||||
4. Build the lifecycle matrix required by the audit plan: first run,
|
||||
already-succeeded skip, self-skip, failure, interruption, forced replacement,
|
||||
non-resumable result, and successful rerun.
|
||||
5. Assign each of the ten cross-boundary scenarios in the audit plan to its
|
||||
primary execution stage and list supporting packages/tests.
|
||||
6. Seed the risk-to-test matrix with the intended test owner for each
|
||||
architectural invariant. Do not judge sufficiency yet.
|
||||
|
||||
### Output
|
||||
|
||||
- Package ownership, stage-contract, lifecycle, scenario, and preliminary
|
||||
risk-to-test matrices.
|
||||
- `ARC` and `RSK` candidates for apparent disagreements, without deciding from
|
||||
documentation alone which artifact is wrong.
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every area in the audit plan's inspection map has an assigned stage.
|
||||
- Every architectural invariant has an implementation owner and intended test
|
||||
owner.
|
||||
- Unknown or contradictory contracts are explicitly recorded.
|
||||
|
||||
## Stage 2: Audit The Runner And Manifest State Machine
|
||||
|
||||
### Entry
|
||||
|
||||
- Stage 1 matrices are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Trace the entry paths into full-run and single-stage execution through
|
||||
`internal/app/planner.go`, `runner.go`, `run_stage.go`, and related helpers.
|
||||
2. Inspect `internal/manifest` models, validation, session/run creation,
|
||||
loading, normalization, atomic saves, and all transition methods.
|
||||
3. Walk every lifecycle-matrix cell through both manifests. Verify clearing
|
||||
versus retention of outputs, diagnostics, generated configuration, metadata,
|
||||
errors, actions, timestamps, and downstream state.
|
||||
4. Reason about failures before and after each session-manifest and run-manifest
|
||||
save. Determine which disagreement states are possible and how a later
|
||||
invocation interprets them.
|
||||
5. Review force, changed-result, self-skip, failed-result, and non-resumable
|
||||
invalidation separately. Confirm behavior at the first and last canonical
|
||||
stage.
|
||||
6. Review session lock acquisition/release and concurrent invocation behavior,
|
||||
while leaving path implementation details to Stage 3.
|
||||
7. Classify the runner's complexity and repeated session/run persistence paths:
|
||||
state-machine clarity, justified explicitness, candidate local helpers, and
|
||||
comments that preserve ordering rationale.
|
||||
8. Review focused app/manifest tests against the matrix and add observations to
|
||||
the risk-to-test ledger.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/app ./internal/manifest
|
||||
go test -race -count=1 ./internal/app ./internal/manifest
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every lifecycle cell has a source-backed conclusion for both manifests.
|
||||
- Cross-boundary scenarios 1, 2, and the lock portion of 10 are resolved or
|
||||
carry explicit questions.
|
||||
- All runner/manifest candidates are confirmed, rejected, or assigned to a
|
||||
named later stage.
|
||||
|
||||
## Stage 3: Audit Paths, Artifacts, And Filesystem Safety
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 1-2 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Review `internal/artifacts`, `internal/artifactpolicy`, `internal/pathsafe`,
|
||||
`internal/fileops`, and local-store filesystem code.
|
||||
2. Inventory canonical path and key helpers, then search callers for ad hoc
|
||||
reconstruction, double normalization, mixed slash/filesystem semantics, or
|
||||
policy implemented outside its owner.
|
||||
3. Trace built-in, configured, extraction, previous-session, and current-state
|
||||
artifact resolution. Verify identity, checksum, contract, provenance,
|
||||
deterministic ordering, and typed missing-state behavior.
|
||||
4. Review atomic file writes, copies, directory promotion, temp cleanup,
|
||||
permission preservation, close/sync/rename errors, existing-destination
|
||||
behavior, same-filesystem assumptions, and platform sensitivity.
|
||||
5. Walk traversal, absolute path, broad root, symlink component, inspected-root
|
||||
replacement, non-regular file, and time-of-check/time-of-use scenarios.
|
||||
6. Confirm that low-level file/storage helpers receive explicit destinations
|
||||
and do not infer stage, campaign, session, run, or publish policy.
|
||||
7. Inspect lock-file implementation and cleanup errors to finish scenario 10.
|
||||
8. Record focused test ownership and gaps without duplicating Stage 2's state
|
||||
conclusions.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/artifacts ./internal/artifactpolicy ./internal/pathsafe ./internal/fileops
|
||||
go test -race -count=1 ./internal/artifacts ./internal/fileops
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every canonical path/key family has one identified owner.
|
||||
- Every material filesystem mutation has documented confinement and atomicity
|
||||
conclusions.
|
||||
- Scenario 10 is resolved.
|
||||
- Safety checks that appear repetitive are classified before any simplification
|
||||
recommendation is made.
|
||||
|
||||
## Stage 4: Audit Publish, Remote Commit, And Cleanup
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 2-3 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Trace publish from stage selection through object-store calls, manifest
|
||||
metadata, commit-marker publication, run completion, and post-publish
|
||||
cleanup.
|
||||
2. Verify prerequisite stage-state checks, selected/configured/extraction
|
||||
output resolution, required versus optional outputs, static and remote
|
||||
locks, run-file exclusions, previous-cache inclusion, and deterministic
|
||||
upload order.
|
||||
3. Enumerate failures before and after every upload. Prove that
|
||||
`current/run_id.txt` is written last and is the only remote-current commit
|
||||
point.
|
||||
4. Review retry/idempotency behavior, existing remote objects, partial uploads,
|
||||
pointer/manifest disagreement, and status/restore interpretation after each
|
||||
partial outcome.
|
||||
5. Trace automatic and manual cleanup gates. Confirm publish execution,
|
||||
`uploaded`, `current_pointer_written`, explicit policy, and confined targets
|
||||
are all required at the correct boundary.
|
||||
6. Confirm that `--force` cannot override publish locks or cleanup safety.
|
||||
7. Review duplication between publish planning, artifact destination policy,
|
||||
operator views, and cleanup metadata only after ownership is established.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/stage ./internal/app ./internal/artifacts ./internal/adapters/storage
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Cross-boundary scenarios 5 and 7 are resolved for every relevant failure
|
||||
boundary.
|
||||
- Remote-current authority and local-cleanup eligibility have explicit truth
|
||||
tables.
|
||||
- Publish findings distinguish stage policy from storage mechanics.
|
||||
|
||||
## Stage 5: Audit Restore And Remote/Previous State
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 2-4 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Trace restore discovery, planning, execution, reporting, audio
|
||||
materialization, and previous-cache planning through `internal/app`,
|
||||
`internal/artifacts`, `internal/previouscache`, `internal/audio`, and storage.
|
||||
2. Confirm remote pointer/manifest identity and campaign/session/run authority,
|
||||
including missing and inconsistent current state.
|
||||
3. Verify remote-to-local confinement, deterministic action ordering,
|
||||
`download`/`skip_same`/`conflict` decisions, force semantics, and dry-run
|
||||
purity.
|
||||
4. Walk failures during download, checksum or manifest validation, atomic
|
||||
install, report persistence, and the manifest-last boundary. Record the
|
||||
intentional lack of rollback and retry consequences.
|
||||
5. Review audio spool/cache identity, cache-hit verification, partial download
|
||||
behavior, and duplicate remote/filesystem work.
|
||||
6. Review previous-session requirement planning, required/optional behavior,
|
||||
identity checks, published-path fallback, and deterministic local mapping.
|
||||
7. Confirm which mechanics are shared with status/validate/operator commands
|
||||
and which caller-specific missing-state policies must remain separate.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/app ./internal/previouscache ./internal/audio ./internal/artifacts ./internal/adapters/storage
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Cross-boundary scenarios 4 and 6 are resolved through retry/recovery.
|
||||
- Restore authority, manifest-last installation, and partial-write behavior are
|
||||
explicit.
|
||||
- Previous-cache conclusions are ready for the prepare and analyze passes.
|
||||
|
||||
## Stage 6: Audit Configuration And Application Composition
|
||||
|
||||
### Entry
|
||||
|
||||
- Stage 1 is complete and Stages 2-5 have identified the policies that
|
||||
configuration and composition must supply.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Review `internal/config`, `cmd/narratio`, application command dispatch,
|
||||
configuration selection, secret-file environment loading, and production
|
||||
collaborator construction.
|
||||
2. Trace discovery, precedence, strict YAML decoding, defaults, empty values,
|
||||
normalization, session templating, and validation order across pipeline,
|
||||
campaign, and session configuration.
|
||||
3. Verify cross-field constraints for stage enablement, paths, timeouts,
|
||||
concurrency, artifacts, Notarius, Scriptorium, publish, storage, cleanup,
|
||||
audio, and previous-session behavior.
|
||||
4. Compare validation logic with maintained examples and the public
|
||||
configuration contract. Record contract drift rather than silently choosing
|
||||
code or docs.
|
||||
5. Check that filesystem secrets are loaded before the boundary that consumes
|
||||
them and are excluded from logs, manifests, reports, generated files, and
|
||||
errors.
|
||||
6. Review conditional construction of expensive/external collaborators and
|
||||
cleanup of anything with a lifecycle. Confirm test injection cannot create a
|
||||
behavior different from production composition.
|
||||
7. Classify repeated validators, path checks, timeout parsing, constructor
|
||||
wrappers, and single-stage command wrappers by policy owner.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/config ./internal/app ./cmd/narratio
|
||||
go vet ./...
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every operator-visible field used by audited behavior has a traced default,
|
||||
normalization, validation, and consumer.
|
||||
- Composition conclusions cover enabled and disabled stages without requiring
|
||||
live services or credentials.
|
||||
- Maintained examples have an explicit validity conclusion.
|
||||
|
||||
## Stage 7: Audit External Adapters And Shared Support
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 3 and 6 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Review `internal/adapters`, `internal/audio`, `internal/logging`,
|
||||
`internal/contracts`, and `internal/artifactmodel` at their public package
|
||||
boundaries.
|
||||
2. For each HTTP, subprocess, notification, and object-storage adapter, compare
|
||||
implementation with its integration contract and trace all production
|
||||
callers.
|
||||
3. Verify context cancellation, timeout ownership, process termination and
|
||||
waiting, goroutine/channel closure, HTTP response-body closure, retries,
|
||||
malformed responses, streaming, pagination, not-found mapping, and local
|
||||
file cleanup.
|
||||
4. Confirm command argument construction, working directory, environment,
|
||||
generated configuration, stdout/stderr separation, output validation, and
|
||||
external error adaptation stay inside the owning adapter.
|
||||
5. Compare subprocess implementations to the shared subprocess package. Classify
|
||||
repeated constructor/config/log/output mechanics separately from
|
||||
adapter-specific protocol policy.
|
||||
6. Review fakes for realistic state and concurrency behavior, but defer their
|
||||
suite-wide value judgment to Stage 12.
|
||||
7. Check shared models for avoidable conversions, stable serialization,
|
||||
validation ownership, and redaction-sensitive diagnostic fields.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/adapters/... ./internal/audio ./internal/logging ./internal/contracts ./internal/artifactmodel
|
||||
go test -race -count=1 ./internal/adapters/... ./internal/audio
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every external resource has an explicit acquisition, cancellation, and
|
||||
release conclusion.
|
||||
- Transport types and protocol policy have not leaked into stages.
|
||||
- Adapter duplication candidates identify the correct shared or specific
|
||||
owner.
|
||||
|
||||
## Stage 8: Audit Prepare And Transcript-Processing Stages
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 2-3 and 6-7 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Review `prepare`, `transcribe`, `merge`, `polish`, `normalize`, `trim`, and
|
||||
`render` as vertical slices from resolved configuration and manifest input
|
||||
through adapter call, run-local output, validation, canonical
|
||||
materialization, and recorded result.
|
||||
2. Verify each implementation against the Stage 1 contract matrix and focused
|
||||
internal document. Record any undeclared input, output, diagnostic, config,
|
||||
adapter, or skip/failure behavior.
|
||||
3. For prepare, confirm local/S3 exclusivity, stable input copying,
|
||||
previous-cache clearing/hydration, and deterministic manifest input records.
|
||||
4. For transcribe, confirm unique speaker identities, bounded runtime
|
||||
concurrency, cancellation, deterministic result ordering, adapter-returned
|
||||
path identity, and partial failure behavior.
|
||||
5. For transformation/render stages, confirm manifest-first resolution,
|
||||
run-local paths, schema/report validation, disabled/default behavior,
|
||||
canonical promotion, and diagnostic-versus-artifact classification.
|
||||
6. Compare similar stage implementations for shared mechanisms only after
|
||||
listing meaningful differences. Avoid a generic stage framework.
|
||||
7. Add stage-focused test ownership, gaps, and redundancy candidates to the
|
||||
risk-to-test matrix.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/stage ./internal/audio ./internal/previouscache ./internal/adapters/whisperx ./internal/adapters/seriatim ./internal/adapters/audita ./internal/adapters/scriptorium
|
||||
go test -race -count=1 ./internal/stage ./internal/audio
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every reviewed stage has a completed contract-matrix row.
|
||||
- Cross-boundary scenario 8 is resolved for transcription and subprocess-backed
|
||||
transformation stages.
|
||||
- Similarity candidates are classified as intentional explicitness, local
|
||||
helper candidates, or shared-owner findings.
|
||||
|
||||
## Stage 9: Audit The Extraction Vertical Slice
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 2-3 and 6-7 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Trace extraction from configuration validation and composition through
|
||||
transcript resolution, invocation fingerprint, Notarius execution, receipt
|
||||
and lane validation, directory promotion, manifest recording, catalog
|
||||
hydration, resume validation, analyze, and publish consumers.
|
||||
2. Verify run-local isolation, regular-file and confined-index requirements,
|
||||
required-lane policy, contract/provenance construction, checksum timing,
|
||||
immutable destination identity, and no-replacement promotion.
|
||||
3. Enumerate failures before and after subprocess completion, receipt parsing,
|
||||
payload inspection, promotion, and manifest persistence. Determine what
|
||||
remains diagnostic, durable, advertised, and reusable.
|
||||
4. Walk every resume validation branch. Distinguish obsolete/missing outcomes
|
||||
that trigger rerun from unsafe conditions that must stop execution.
|
||||
5. Evaluate the fingerprint's intentionally observable and unobservable inputs
|
||||
against documentation and force guidance.
|
||||
6. Review the dense validation code for named sub-decisions and comments while
|
||||
preserving the visible security proof and check ordering.
|
||||
7. Confirm focused tests cover immediate reuse, cross-invocation reuse,
|
||||
configuration change, payload tampering, provenance mismatch, symlinks/root
|
||||
replacement, failure residue, and downstream invalidation at the correct
|
||||
layers.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/stage ./internal/artifacts ./internal/fileops ./internal/adapters/notarius ./internal/app
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Cross-boundary scenario 3 is resolved, including transitive-input limits.
|
||||
- Promotion, advertisement, and resume each have a distinct authority and
|
||||
failure conclusion.
|
||||
- Every proposed simplification states which security or compatibility checks
|
||||
it preserves.
|
||||
|
||||
## Stage 10: Audit Analyze And Artifact Dependencies
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 3, 6, 8, and 9 are complete.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Trace all analyze source families from configuration validation through
|
||||
runtime catalog registration, availability, resolution, Scriptorium
|
||||
execution/reuse, materialization, metadata, and publish selection.
|
||||
2. Verify enabled, selected, executable, reused, generated, and unavailable
|
||||
states are distinct and deterministic.
|
||||
3. Review configured-artifact dependency validation and runtime topological
|
||||
ordering for cycles, missing dependencies, stable ordering, and consistency
|
||||
between configuration and execution.
|
||||
4. Confirm required/optional behavior and guidance for built-in transcripts,
|
||||
prepared stable inputs, configured artifacts, extraction sources, and
|
||||
previous-session sources.
|
||||
5. Prove previous-session resolution is local-only during analyze and that
|
||||
disabled artifacts are reused only under the documented conditions.
|
||||
6. Inspect repeated resolution branches, parameter width, nested lookup, and
|
||||
ordering work for a smaller representation or indexed plan without merging
|
||||
distinct source policies.
|
||||
7. Review tests for each state transition and source family at the narrowest
|
||||
stable owner, noting semantic duplication across config, artifacts, stage,
|
||||
publish, and assembled runner tests.
|
||||
|
||||
### Validation
|
||||
|
||||
```sh
|
||||
go test -count=1 ./internal/stage ./internal/artifacts ./internal/artifactpolicy ./internal/config ./internal/adapters/scriptorium ./internal/app
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Cross-boundary scenario 9 is resolved for every source family and selection
|
||||
state.
|
||||
- Dependency ordering and source availability have explicit determinism and
|
||||
complexity conclusions.
|
||||
- Config, artifact-policy, catalog, stage, and publish ownership is unambiguous
|
||||
or represented by an `ARC` finding.
|
||||
|
||||
## Stage 11: Audit Duplication, Simplicity, Efficiency, And Comments
|
||||
|
||||
### Entry
|
||||
|
||||
- Behavior stages 2-10 are complete, so structural candidates can be judged
|
||||
against known contracts.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Rerun graph similarity, complexity, fan-in/fan-out, call-path, loop-depth,
|
||||
scan-in-loop, and change-coupling analyses on production code. Add targeted
|
||||
text/static searches for patterns the graph cannot represent.
|
||||
2. Revisit all `DUP`, `SIM`, `EFF`, and `COM` candidates collected earlier.
|
||||
Search for additional occurrences and trace all callers before assigning an
|
||||
owner.
|
||||
3. For duplication, classify coincidental syntax, shared mechanism, duplicated
|
||||
policy, or deliberately explicit security/state logic. Propose only the
|
||||
narrowest helper that improves ownership and comprehension.
|
||||
4. For complexity, sketch the smaller control flow or data model and verify it
|
||||
leaves state transitions, validation order, and commit boundaries visible.
|
||||
5. For efficiency, state the input scale or call frequency, current and proposed
|
||||
complexity/I/O behavior, expected benefit, and benchmark or measurement
|
||||
needed. Reject micro-optimizations without a credible workload.
|
||||
6. Review standard-library usage, errors, slices/maps, allocations, copying,
|
||||
sorting, serialization, filesystem passes, adapter initialization, remote
|
||||
calls, goroutines/channels, and interface breadth across the complete codebase.
|
||||
7. Review comments only after simplification decisions. Recommend why-comments
|
||||
for remaining invariants, compatibility limits, safety checks, partial
|
||||
failure, and ordering; flag comments that restate code or no longer match it.
|
||||
8. Check dependencies and platform assumptions for clear correctness,
|
||||
portability, complexity, or maintenance consequences.
|
||||
|
||||
### Validation
|
||||
|
||||
- Run focused package tests for any behavior used to disprove or confirm a
|
||||
candidate.
|
||||
- Run existing benchmarks where relevant. Propose a benchmark rather than
|
||||
inventing performance claims when representative measurement is absent.
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every structural candidate is confirmed, rejected with a reason, or merged
|
||||
into a stronger root-cause finding.
|
||||
- No helper recommendation creates a generic workflow abstraction or moves
|
||||
policy into a low-level utility.
|
||||
- Every efficiency finding has a credible workload and validation method.
|
||||
- Every comment finding states the non-obvious rationale that should be
|
||||
preserved.
|
||||
|
||||
## Stage 12: Audit The Test Suite Against Policy
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 2-11 have populated the risk-to-test matrix and test observations.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Complete the risk-to-test matrix. For every consequential invariant, list
|
||||
the current tests, proper owner, protected defect, missing failure modes, and
|
||||
overlap with other layers.
|
||||
2. Review tests by behavior cluster rather than filename: parsing/validation,
|
||||
domain/state, filesystem, adapters, orchestration, CLI, integration, and
|
||||
representative assembled workflows.
|
||||
3. Classify gaps for data integrity, destructive operations, compatibility,
|
||||
security, concurrency, idempotency, recovery, cancellation, and partial
|
||||
success. Confirm the gap is not credibly protected elsewhere.
|
||||
4. Classify redundancy and brittleness: private constants/defaults, exact error
|
||||
wording, incidental formatting/paths, mock choreography, oversized
|
||||
snapshots, helper-level duplication, and the same policy repeated across
|
||||
layers.
|
||||
5. Review doubles using the policy order: real deterministic collaborator,
|
||||
stateful fake, stub, then mock when interaction is contractual. Check that
|
||||
fakes model the failure and state semantics used by the tests.
|
||||
6. Inspect test helpers and large test functions for simplification and
|
||||
meaningful table-driven boundaries without creating a fixture framework
|
||||
whose maintenance cost exceeds its value.
|
||||
7. Review determinism and isolation: credentials, network access, paid APIs,
|
||||
environment, working directory, clocks, randomness, ports, temp paths,
|
||||
process-global state, ordering, cleanup, and parallel execution.
|
||||
8. Use coverage to investigate consequential weak branches, not as a score.
|
||||
Review heavily covered behavior for marginal-value duplication as well.
|
||||
9. Identify focused fuzz opportunities for parsers, YAML/JSON normalization,
|
||||
source IDs, confined paths, remote/local mapping, and manifest decoding.
|
||||
10. Compare local requirements with `.woodpecker/` and other automation. Record
|
||||
missing enforcement as a risk/cost decision, not an assumption that every
|
||||
diagnostic command belongs in CI.
|
||||
11. Investigate order dependence and flakiness with bounded runs, recording
|
||||
runtime and any reproducible seed:
|
||||
|
||||
```sh
|
||||
go test -shuffle=on -count=3 ./...
|
||||
go test -race -shuffle=on -count=1 ./...
|
||||
```
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every important risk has a sufficiency conclusion and one intended test
|
||||
owner.
|
||||
- Every proposed test addition names the realistic defect and marginal value.
|
||||
- Every deletion/consolidation names the stronger remaining protection.
|
||||
- Default-suite determinism, offline behavior, runtime, flakiness, and CI
|
||||
enforcement have explicit conclusions.
|
||||
|
||||
## Stage 13: Synthesize And Close The Audit
|
||||
|
||||
### Entry
|
||||
|
||||
- Stages 0-12 meet their exit gates or have explicitly accepted limitations.
|
||||
|
||||
### Execute
|
||||
|
||||
1. Reconcile candidates and findings across stages. Merge shared root causes and
|
||||
remove repeated symptoms while retaining all affected locations and
|
||||
contracts.
|
||||
2. Recheck every confirmed finding against current source, callers, tests, and
|
||||
canonical documentation. Downgrade or reject anything supported only by a
|
||||
metric or hypothetical preference.
|
||||
3. Rank impact, likelihood, confidence, and remediation scope separately. Order
|
||||
the recommended backlog by dependency: correctness/data safety first,
|
||||
architectural ownership next, then simplification/duplication, tests,
|
||||
efficiency, and comments where they remain necessary.
|
||||
4. Record positive conclusions for high-risk areas where the current design and
|
||||
tests are sufficient. The report should not imply that only defective areas
|
||||
were reviewed.
|
||||
5. Reconcile the area coverage ledger, lifecycle matrix, cross-boundary scenario
|
||||
matrix, and risk-to-test matrix with the audit plan's completion criteria.
|
||||
6. Record any accepted risks, ambiguous contracts, environmental limitations,
|
||||
and deferred investigations with an explicit rationale and owner.
|
||||
7. Check whether HEAD or the worktree changed since Stage 0. Rerun affected
|
||||
stages or clearly pin the report to the original revision.
|
||||
8. Validate the report and roadmap document links and run `git diff --check`.
|
||||
If implementation changed during the audit, rerun the full Stage 0 validation
|
||||
baseline against the final audited revision.
|
||||
|
||||
### Final Deliverable
|
||||
|
||||
`docs/roadmap/audit-findings.md` must contain:
|
||||
|
||||
- an executive assessment without unsupported quality scores;
|
||||
- the audited revision and validation baseline;
|
||||
- coverage and scenario completion summaries;
|
||||
- confirmed findings ordered by dependency and risk;
|
||||
- rejected candidate themes where their recurrence would otherwise waste work;
|
||||
- the test-suite sufficiency assessment;
|
||||
- positive conclusions and accepted risks; and
|
||||
- a recommended remediation order, without implementing the remediation.
|
||||
|
||||
### Exit Gate
|
||||
|
||||
- Every completion criterion in the audit plan is satisfied or explicitly
|
||||
marked limited with rationale.
|
||||
- Every finding is evidence-backed, deduplicated, actionable, and assigned a
|
||||
stable ID.
|
||||
- No production change is included in the audit output.
|
||||
- The report is sufficient to prepare a separate remediation sequence without
|
||||
repeating discovery.
|
||||
@@ -117,6 +117,40 @@ Safe fix:
|
||||
|
||||
Relevant reference: [CLI artifact selection](./cli.md).
|
||||
|
||||
## Bounded run prerequisite is unusable
|
||||
|
||||
Symptom:
|
||||
|
||||
- `run` or `session plan` reports that a prerequisite stage is absent or has a
|
||||
pending, running, failed, stale, or interrupted status before the selected
|
||||
start.
|
||||
|
||||
Likely cause:
|
||||
|
||||
- `--from` excludes upstream work that has not reached the terminal
|
||||
`succeeded` or `skipped` state in the session manifest.
|
||||
|
||||
Diagnostics:
|
||||
|
||||
```bash
|
||||
narratio session status 2026-04-04
|
||||
narratio session plan 2026-04-04 --from render --through analyze
|
||||
```
|
||||
|
||||
Safe fix:
|
||||
|
||||
- widen the bounded range to include the first reported stage, or recover that
|
||||
stage explicitly with `run-stage` before retrying. The failed check does not
|
||||
create a run record or modify the manifest. Narratio does not resume-validate
|
||||
excluded prefix stages, and stages after `--through` are not prerequisites.
|
||||
|
||||
If prerequisite statuses are terminal but a selected stage reports a missing,
|
||||
unsafe, or checksum-inconsistent artifact, repair the artifact at the stage
|
||||
that owns it; do not edit the manifest to bypass the selected stage's concrete
|
||||
input validation.
|
||||
|
||||
Relevant reference: [Operations: Stage Execution and Continuation Behavior](./operations.md#stage-execution-and-continuation-behavior).
|
||||
|
||||
## Notarius executable missing
|
||||
|
||||
Symptom:
|
||||
@@ -158,6 +192,66 @@ is expected audit state, not a signal to relink the old bundle manually.
|
||||
|
||||
Relevant reference: [Operations: Extraction Workflow](./operations.md#extraction-workflow).
|
||||
|
||||
## Prepared Notarius reference missing or inconsistent
|
||||
|
||||
Symptom:
|
||||
|
||||
- extraction or resume validation reports that a configured reference source is
|
||||
unavailable, unsafe, empty, or checksum-inconsistent and recommends
|
||||
`prepare --force`.
|
||||
|
||||
Likely causes:
|
||||
|
||||
- `prepare` has not run since the campaign/session stable input changed;
|
||||
- the configured source file is missing;
|
||||
- a prepared `inputs/` file or its manifest record was modified independently;
|
||||
- a spell-catalog binding exists without an effective `spell_catalog_file`.
|
||||
|
||||
Diagnostics:
|
||||
|
||||
```bash
|
||||
narratio session status 2026-04-04
|
||||
narratio session validate 2026-04-04
|
||||
```
|
||||
|
||||
Safe fix:
|
||||
|
||||
- correct the campaign/session input path, then refresh canonical prepared
|
||||
evidence before extraction:
|
||||
|
||||
```bash
|
||||
narratio run-stage prepare 2026-04-04 --force
|
||||
```
|
||||
|
||||
Do not point Notarius directly at the original source path or edit the manifest
|
||||
checksum. Relevant references: [Notarius reference configuration](./config.md#notarius-reference-bindings)
|
||||
and [Operations: Extraction Workflow](./operations.md#extraction-workflow).
|
||||
|
||||
## Notarius reference selector or generated-handoff collision
|
||||
|
||||
Symptom:
|
||||
|
||||
- Notarius exits nonzero with an undeclared reference-slot, incompatible media,
|
||||
or external/generated reference collision error.
|
||||
|
||||
Likely causes:
|
||||
|
||||
- a selector does not identify a slot declared by the selected Notarius target;
|
||||
- a prepared file does not satisfy that slot's Notarius media contract; or
|
||||
- a CLI binding attempts to replace a same-run generated D&D handoff.
|
||||
|
||||
Safe fix:
|
||||
|
||||
- compare external bindings with the selected Notarius pipeline's canonical
|
||||
consumer documentation;
|
||||
- keep only campaign-owned external slots on the CLI; and
|
||||
- leave registry, scene, combat, and occurrence handoffs to Notarius pipeline
|
||||
composition.
|
||||
|
||||
Narratio validates selector structure and prepared evidence, while Notarius
|
||||
owns slot declarations, media compatibility, and generated-handoff conflicts.
|
||||
Relevant reference: [Notarius integration](./integrations/notarius.md).
|
||||
|
||||
## Atomic Notarius promotion unsupported
|
||||
|
||||
Symptom:
|
||||
@@ -198,7 +292,7 @@ Safe fix:
|
||||
|
||||
- compare installed Notarius output with the canonical Notarius contracts,
|
||||
including receipt `index_file: index.json` and index management names
|
||||
`manifest.json`, `rejected.json`, and `warnings.json`; align
|
||||
`manifest.json`, `rejected.json`, `warnings.json`, and `diagnostics.json`; align
|
||||
`pipeline.notarius` constraints and rerun. Do not bypass confinement or schema
|
||||
checks.
|
||||
|
||||
@@ -230,6 +324,8 @@ Likely causes:
|
||||
|
||||
- the executable/config path, pipeline ID, timeout, working directory, or
|
||||
configured output contracts changed;
|
||||
- a configured prepared reference selector, source, path, checksum, or byte
|
||||
size changed;
|
||||
- the durable bundle, index, lane set, provenance, regular-file status, or
|
||||
checksum no longer validates.
|
||||
|
||||
@@ -252,12 +348,110 @@ Safe fix:
|
||||
narratio run-stage extract 2026-04-04 --force
|
||||
```
|
||||
|
||||
Narratio fingerprints its invocation contract, not the contents of transitive
|
||||
Notarius inputs. Always force extraction after changing them; downstream
|
||||
Narratio fingerprints its invocation contract and prepared Narratio reference
|
||||
identities, not the contents of other transitive Notarius inputs. Always force
|
||||
extraction after changing those external inputs; downstream
|
||||
successful stages are then marked stale normally.
|
||||
|
||||
Relevant reference: [Operations: Extraction Workflow](./operations.md#extraction-workflow).
|
||||
|
||||
## Analysis artifact evidence is not current
|
||||
|
||||
Symptom:
|
||||
|
||||
- ordinary continuation or `session plan` schedules one or more configured
|
||||
artifacts even though a canonical output file exists; or
|
||||
- publish reports a configured artifact source unavailable.
|
||||
|
||||
Likely causes:
|
||||
|
||||
- the per-artifact record is stale, missing, failed, unselected, malformed, or
|
||||
from the legacy aggregate-only manifest contract;
|
||||
- a configured prompt/profile, dependency, input identity, output path, or
|
||||
effective variable changed; or
|
||||
- the recorded output is missing, unsafe, empty, or has a size/checksum that no
|
||||
longer matches its manifest evidence.
|
||||
|
||||
Diagnostics:
|
||||
|
||||
```bash
|
||||
narratio session status 2026-04-04
|
||||
narratio session artifacts 2026-04-04
|
||||
narratio session plan 2026-04-04 --from analyze --through analyze
|
||||
```
|
||||
|
||||
Safe fix:
|
||||
|
||||
- investigate unexpected path or checksum changes as possible tampering;
|
||||
- otherwise let the selected analyze work rerun, or explicitly regenerate only
|
||||
the affected targets; and
|
||||
- never edit the fingerprint/checksum in the manifest or copy an old file into
|
||||
the canonical path as a substitute for current evidence.
|
||||
|
||||
```bash
|
||||
narratio analyze 2026-04-04 --artifacts session_recap
|
||||
```
|
||||
|
||||
Relevant references: [Operations: Artifact Selection](./operations.md#artifact-selection)
|
||||
and [Artifact Internals](./internal/artifacts.md#resolution-rules).
|
||||
|
||||
## Legacy aggregate analysis requires regeneration
|
||||
|
||||
Symptom:
|
||||
|
||||
- a manifest from an older Narratio version reports aggregate analyze success
|
||||
and the old files are present, but configured artifact sources remain
|
||||
unavailable.
|
||||
|
||||
Likely cause:
|
||||
|
||||
- the manifest has no supported per-artifact analyze state. Aggregate output
|
||||
lists do not establish current configured-artifact authority.
|
||||
|
||||
Safe fix:
|
||||
|
||||
- regenerate the required artifacts. A partial selection makes only its
|
||||
targets and prerequisites eligible for current state; unselected legacy
|
||||
files intentionally remain unavailable. Run full analysis later when every
|
||||
enabled configured artifact must become current.
|
||||
|
||||
```bash
|
||||
narratio analyze 2026-04-04 --artifacts session_recap
|
||||
narratio analyze 2026-04-04
|
||||
```
|
||||
|
||||
After current records exist, inspect them and publish explicitly. Do not delete
|
||||
the legacy files merely to influence selection; availability is manifest-owned.
|
||||
|
||||
Relevant references: [Operations: Stage Execution and Continuation Behavior](./operations.md#stage-execution-and-continuation-behavior)
|
||||
and [Manifest Internals](./internal/manifest.md#analyze-owned-artifact-state).
|
||||
|
||||
## Scriptorium private input changed without a rerun
|
||||
|
||||
Symptom:
|
||||
|
||||
- a prompt, profile, imported configuration file, executable, or other input
|
||||
loaded privately by Scriptorium changed, but Narratio still considers an
|
||||
artifact current.
|
||||
|
||||
Likely cause:
|
||||
|
||||
- analysis fingerprints cover Narratio-observable semantic identities, not
|
||||
executable contents or arbitrary files and transitive configuration that
|
||||
Scriptorium loads behind its configured paths and identifiers.
|
||||
|
||||
Safe fix:
|
||||
|
||||
- explicitly force the affected target after changing an unobserved private
|
||||
input. Force applies to explicit targets; current prerequisites remain
|
||||
reusable unless selected themselves.
|
||||
|
||||
```bash
|
||||
narratio analyze 2026-04-04 --artifacts session_recap
|
||||
```
|
||||
|
||||
Relevant reference: [Analyze Internals](./internal/stage-analyze.md#invariants).
|
||||
|
||||
## Previous-session artifact input missing
|
||||
|
||||
Symptom:
|
||||
@@ -299,7 +493,7 @@ Symptom:
|
||||
Likely causes:
|
||||
|
||||
- another process is running for the same session;
|
||||
- stale lock left by interrupted process.
|
||||
- a process still holds the operating-system lock while it is shutting down.
|
||||
|
||||
Diagnostics:
|
||||
|
||||
@@ -311,7 +505,8 @@ ps aux | grep narratio
|
||||
Safe fix:
|
||||
|
||||
- wait for active process completion;
|
||||
- remove stale lock only after confirming no live process owns it.
|
||||
- retry after an interrupted holder has exited; the kernel releases its lock
|
||||
even though the `.lock` metadata file remains for inspection.
|
||||
|
||||
Relevant reference: [Operations: Local State Layout](./operations.md#local-state-layout).
|
||||
|
||||
@@ -435,7 +630,7 @@ Diagnostics:
|
||||
|
||||
```bash
|
||||
ls -la /path/to/secrets_dir
|
||||
env | grep -E 'OBJECT_STORAGE|AWS|AUDITA|SCRIPTORIUM'
|
||||
env | sed 's/=.*//' | grep -E 'OBJECT_STORAGE|AWS|AUDITA|SCRIPTORIUM'
|
||||
```
|
||||
|
||||
Safe fix:
|
||||
|
||||
@@ -39,7 +39,9 @@ with the sample campaign and a compatible local- or S3-audio session.
|
||||
[autocorrect](campaigns/sample-campaign/autocorrect.yml),
|
||||
[glossary](campaigns/sample-campaign/glossary.yml),
|
||||
[players](campaigns/sample-campaign/players.yml), and
|
||||
[party](campaigns/sample-campaign/party.yml) fixtures.
|
||||
[party](campaigns/sample-campaign/party.yml) fixtures, plus an optional
|
||||
[spell-catalog overlay](campaigns/sample-campaign/spell_catalog.json) that
|
||||
follows the Notarius v0.6 contract.
|
||||
- [Sample speaker audio](audio/sample-speaker.flac) is a text placeholder that
|
||||
reserves the expected filename and directory shape. Replace it with a real
|
||||
FLAC file before running transcription.
|
||||
|
||||
@@ -6,3 +6,4 @@ inputs:
|
||||
glossary_file: ./glossary.yml
|
||||
players_file: ./players.yml
|
||||
party_file: ./party.yml
|
||||
spell_catalog_file: ./spell_catalog.json
|
||||
|
||||
18
examples/campaigns/sample-campaign/spell_catalog.json
Normal file
18
examples/campaigns/sample-campaign/spell_catalog.json
Normal file
@@ -0,0 +1,18 @@
|
||||
{
|
||||
"schema_version": "notarius.dnd.spell-catalog-overlay.v1",
|
||||
"catalogs": [
|
||||
{
|
||||
"id": "narratio.sample-campaign",
|
||||
"ruleset": "dnd-5e-2014",
|
||||
"source": {
|
||||
"title": "Narratio sample campaign spell names"
|
||||
},
|
||||
"spells": [
|
||||
{
|
||||
"name": "Aegis of Emberfall",
|
||||
"aliases": ["Emberfall Aegis"]
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -14,6 +14,11 @@ notarius:
|
||||
config_path: /usr/local/etc/notarius/config.yml
|
||||
pipeline_id: dnd-session
|
||||
timeout: 3h
|
||||
references:
|
||||
glossary: narratio.input.glossary
|
||||
party: narratio.input.party
|
||||
players: narratio.input.players
|
||||
spell_catalog: narratio.input.spell_catalog
|
||||
outputs:
|
||||
npc_registry:
|
||||
lane_id: npc-registry
|
||||
@@ -52,4 +57,3 @@ scriptorium:
|
||||
scenes:
|
||||
source: narratio.extraction.scene_descriptions
|
||||
required: true
|
||||
|
||||
|
||||
@@ -12,7 +12,7 @@ workspace:
|
||||
# env_dir: ./secrets
|
||||
|
||||
storage:
|
||||
# Optional storage backend selector; use "s3" for publish + S3 audio workflows.
|
||||
# Defaults to "local". Use "s3" explicitly for publish + S3 audio workflows.
|
||||
backend: s3
|
||||
s3:
|
||||
# Required when using S3 audio or S3 publish uploads.
|
||||
@@ -136,6 +136,13 @@ notarius:
|
||||
pipeline_id: dnd-session
|
||||
timeout: 3h
|
||||
working_directory: /usr/local/etc/notarius
|
||||
# External campaign references use prepared Narratio source IDs. Omit an
|
||||
# optional binding when the selected Notarius pipeline does not need it.
|
||||
references:
|
||||
glossary: narratio.input.glossary
|
||||
party: narratio.input.party
|
||||
players: narratio.input.players
|
||||
spell_catalog: narratio.input.spell_catalog
|
||||
# Each key creates source narratio.extraction.<key>. These constraints match
|
||||
# the current Notarius D&D lane contracts; update them with Notarius.
|
||||
outputs:
|
||||
@@ -260,7 +267,5 @@ scriptorium:
|
||||
output_kind: player_handout
|
||||
|
||||
notification:
|
||||
# Optional notification settings.
|
||||
backend: ""
|
||||
recipient: ""
|
||||
timeout: 30s
|
||||
# No delivery provider is currently implemented.
|
||||
mode: noop
|
||||
|
||||
@@ -125,4 +125,4 @@ scriptorium:
|
||||
output_kind: player_handout
|
||||
|
||||
notification:
|
||||
timeout: 30s
|
||||
mode: noop
|
||||
|
||||
@@ -4,10 +4,10 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
// NoopRunner is a deterministic no-op audita adapter.
|
||||
@@ -84,17 +84,17 @@ func materializePlaceholders(req PolishRequest) error {
|
||||
"merged_transcript_path": req.MergedTranscriptPath,
|
||||
"output_path": req.OutputProcessedPath,
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("audita noop/fake stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("audita noop/fake stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("audita noop/fake stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("audita noop/fake stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
@@ -117,14 +117,14 @@ func writeJSONIfRequested(path string, payload any) error {
|
||||
if path == "" {
|
||||
return nil
|
||||
}
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
if err := fileops.EnsureWorkspaceDirectory(filepath.Dir(path)); err != nil {
|
||||
return fmt.Errorf("create parent directory %q: %w", filepath.Dir(path), err)
|
||||
}
|
||||
data, err := json.Marshal(payload)
|
||||
if err != nil {
|
||||
return fmt.Errorf("marshal placeholder json for %q: %w", path, err)
|
||||
}
|
||||
if err := subprocess.WriteFileAtomic(path, data, 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(path, data, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write placeholder json %q: %w", path, err)
|
||||
}
|
||||
return nil
|
||||
|
||||
@@ -6,8 +6,6 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
// TODO: implement a real Audita subprocess/service adapter.
|
||||
|
||||
// Runner is the adapter boundary for audita polish invocations.
|
||||
type Runner interface {
|
||||
Run(ctx context.Context, req PolishRequest) (PolishResult, error)
|
||||
@@ -15,25 +13,15 @@ type Runner interface {
|
||||
|
||||
// PolishRequest describes an audita invocation.
|
||||
type PolishRequest struct {
|
||||
GeneratedConfigPath string
|
||||
MergedTranscriptPath string
|
||||
OutputProcessedPath string
|
||||
GlossaryPath string
|
||||
ReportPath string
|
||||
WorkDir string
|
||||
Modules []string
|
||||
BaseURL string
|
||||
Model string
|
||||
TranscriptDescription string
|
||||
ConfigPath string
|
||||
OutputSchema string
|
||||
WorkDirRetention string
|
||||
TotalLLMConcurrency *int
|
||||
ProposalLLMConcurrency *int
|
||||
ValidationModel string
|
||||
ValidationLLMConcurrency *int
|
||||
StdoutLogPath string
|
||||
StderrLogPath string
|
||||
GeneratedConfigPath string
|
||||
MergedTranscriptPath string
|
||||
OutputProcessedPath string
|
||||
GlossaryPath string
|
||||
ReportPath string
|
||||
WorkDir string
|
||||
Modules []string
|
||||
StdoutLogPath string
|
||||
StderrLogPath string
|
||||
}
|
||||
|
||||
// PolishResult describes a polish output.
|
||||
|
||||
@@ -11,8 +11,15 @@ import (
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
// MaxProcessedOutputBytes bounds Audita's processed-transcript JSON result.
|
||||
const MaxProcessedOutputBytes int64 = 64 * 1024 * 1024
|
||||
|
||||
// MaxReportOutputBytes bounds Audita's optional report JSON result.
|
||||
const MaxReportOutputBytes int64 = 16 * 1024 * 1024
|
||||
|
||||
// SubprocessRunnerConfig defines deterministic settings for Audita CLI execution.
|
||||
type SubprocessRunnerConfig struct {
|
||||
Binary string
|
||||
@@ -207,12 +214,13 @@ func (r *SubprocessRunner) Run(ctx context.Context, req PolishRequest) (PolishRe
|
||||
}
|
||||
|
||||
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
||||
Executable: r.binary,
|
||||
Args: args,
|
||||
Timeout: r.timeout,
|
||||
EnvOverrides: env,
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
Executable: r.binary,
|
||||
Args: args,
|
||||
Timeout: r.timeout,
|
||||
EnvOverrides: env,
|
||||
DiagnosticOwner: "audita",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
if err != nil {
|
||||
wrappedMessage := fmt.Sprintf(
|
||||
@@ -370,13 +378,13 @@ func (r *SubprocessRunner) writeInvocationConfig(req PolishRequest, args []strin
|
||||
"credential_env_var": r.llmAPIKeyEnv,
|
||||
"credential_present": credentialPresent,
|
||||
}
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
func validateProcessedOutput(path string) error {
|
||||
data, err := os.ReadFile(path)
|
||||
data, err := readAuditaResult(path, MaxProcessedOutputBytes, "processed transcript")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read file: %w", err)
|
||||
return err
|
||||
}
|
||||
|
||||
var payload map[string]any
|
||||
@@ -405,9 +413,9 @@ func addSubprocessStreamHint(message string, runErr error) string {
|
||||
}
|
||||
|
||||
func validateJSONFile(path string) error {
|
||||
data, err := os.ReadFile(path)
|
||||
data, err := readAuditaResult(path, MaxReportOutputBytes, "report")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read file: %w", err)
|
||||
return err
|
||||
}
|
||||
var v any
|
||||
if err := json.Unmarshal(data, &v); err != nil {
|
||||
@@ -415,3 +423,11 @@ func validateJSONFile(path string) error {
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func readAuditaResult(path string, limit int64, category string) ([]byte, error) {
|
||||
data, err := fileops.ReadRegularFile(path, limit)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("audita %s result exceeds or cannot be read within %d-byte limit: %w", category, limit, err)
|
||||
}
|
||||
return data, nil
|
||||
}
|
||||
|
||||
@@ -189,7 +189,7 @@ func TestSubprocessRunnerUnconfiguredCredentialEnvOmitsCredential(t *testing.T)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSubprocessRunnerInheritsParentEnvironment(t *testing.T) {
|
||||
func TestSubprocessRunnerOmitsUnspecifiedParentEnvironment(t *testing.T) {
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("helper wrapper script uses /bin/sh")
|
||||
}
|
||||
@@ -214,8 +214,8 @@ func TestSubprocessRunnerInheritsParentEnvironment(t *testing.T) {
|
||||
}
|
||||
|
||||
rec := readAuditaHelperRecord(t, recordPath)
|
||||
if rec.Env["AUDITA_INHERITED_MARKER"] != "inherited-from-parent" {
|
||||
t.Fatalf("AUDITA_INHERITED_MARKER = %q, want inherited-from-parent", rec.Env["AUDITA_INHERITED_MARKER"])
|
||||
if rec.Env["AUDITA_INHERITED_MARKER"] != "" {
|
||||
t.Fatalf("AUDITA_INHERITED_MARKER = %q, want omitted from the child environment", rec.Env["AUDITA_INHERITED_MARKER"])
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -14,7 +14,9 @@ func (f *FakeRunner) Run(ctx context.Context, req RunRequest) (RunResult, error)
|
||||
if err := ctx.Err(); err != nil {
|
||||
return RunResult{}, err
|
||||
}
|
||||
f.Requests = append(f.Requests, req)
|
||||
copyRequest := req
|
||||
copyRequest.References = append([]ReferenceBinding(nil), req.References...)
|
||||
f.Requests = append(f.Requests, copyRequest)
|
||||
if f.Err != nil {
|
||||
return RunResult{}, f.Err
|
||||
}
|
||||
|
||||
@@ -6,13 +6,20 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
const ReceiptSchemaVersion = "notarius.run-result.v1"
|
||||
const ReceiptSchemaVersion = "notarius.run-result.v2"
|
||||
|
||||
// Runner is the adapter boundary for a complete Notarius pipeline invocation.
|
||||
type Runner interface {
|
||||
Run(ctx context.Context, req RunRequest) (RunResult, error)
|
||||
}
|
||||
|
||||
// ReferenceBinding maps one normalized Notarius selector to an absolute
|
||||
// external reference path.
|
||||
type ReferenceBinding struct {
|
||||
Selector string
|
||||
Path string
|
||||
}
|
||||
|
||||
// RunRequest contains the resolved inputs and diagnostic destinations for one invocation.
|
||||
type RunRequest struct {
|
||||
Binary string
|
||||
@@ -24,20 +31,41 @@ type RunRequest struct {
|
||||
ReceiptPath string
|
||||
LogPath string
|
||||
Timeout time.Duration
|
||||
References []ReferenceBinding
|
||||
}
|
||||
|
||||
// Receipt is the transport-neutral successful run receipt.
|
||||
type Receipt struct {
|
||||
SchemaVersion string
|
||||
RunID string
|
||||
PipelineID string
|
||||
OutputDirectory string
|
||||
IndexFile string
|
||||
NormalizedOutputCount int
|
||||
RejectedOutputCount int
|
||||
WarningCount int
|
||||
ValidationStatus string
|
||||
DebugDirectory string
|
||||
SchemaVersion string
|
||||
RunID string
|
||||
PipelineID string
|
||||
OutputDirectory string
|
||||
IndexFile string
|
||||
NormalizedOutputCount int
|
||||
RejectedOutputCount int
|
||||
WarningGroupCount int
|
||||
WarningOccurrenceCount int
|
||||
DiagnosticGroupCount int
|
||||
DiagnosticOccurrenceCount int
|
||||
DiagnosticsTruncated bool
|
||||
ValidationStatus string
|
||||
ValidationSummaries []ValidationSummary
|
||||
DebugDirectory string
|
||||
}
|
||||
|
||||
// ValidationSummary retains the bounded outcome of one Notarius producer result.
|
||||
type ValidationSummary struct {
|
||||
Stage string
|
||||
StepID string
|
||||
LaneID string
|
||||
ModuleKey string
|
||||
ChunkID string
|
||||
Status string
|
||||
RejectingValidators []string
|
||||
ReasonCodes []string
|
||||
IncompleteValidators []string
|
||||
ProducerAttemptCount int
|
||||
TerminalAction string
|
||||
}
|
||||
|
||||
// LaneDescriptor identifies one normalized lane payload discovered through the index.
|
||||
@@ -72,6 +100,8 @@ type Index struct {
|
||||
RejectedPath string
|
||||
WarningsFile string
|
||||
WarningsPath string
|
||||
DiagnosticsFile string
|
||||
DiagnosticsPath string
|
||||
Lanes []LaneDescriptor
|
||||
ChunkMap *PipelineDescriptor
|
||||
EvidenceContext *PipelineDescriptor
|
||||
@@ -90,8 +120,29 @@ type RejectionSummary struct {
|
||||
|
||||
// WarningSummary retains structured warning identity without free-form messages.
|
||||
type WarningSummary struct {
|
||||
Scope string
|
||||
ReasonCode string
|
||||
Disposition string
|
||||
Category string
|
||||
ReasonCode string
|
||||
Origin DiagnosticOrigin
|
||||
OccurrenceCount int
|
||||
}
|
||||
|
||||
// DiagnosticOrigin identifies the framework-owned pipeline location of a finding.
|
||||
type DiagnosticOrigin struct {
|
||||
Stage string
|
||||
StepID string
|
||||
LaneID string
|
||||
ModuleKey string
|
||||
ValidatorKey string
|
||||
}
|
||||
|
||||
// DiagnosticSummary retains bounded advisory or observation group metadata.
|
||||
type DiagnosticSummary struct {
|
||||
Disposition string
|
||||
Category string
|
||||
ReasonCode string
|
||||
Origin DiagnosticOrigin
|
||||
OccurrenceCount int
|
||||
}
|
||||
|
||||
// RunResult describes a successfully decoded and validated Notarius bundle.
|
||||
@@ -105,4 +156,5 @@ type RunResult struct {
|
||||
Duration time.Duration
|
||||
Rejections []RejectionSummary
|
||||
Warnings []WarningSummary
|
||||
Diagnostics []DiagnosticSummary
|
||||
}
|
||||
|
||||
@@ -5,23 +5,30 @@ import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/notariusref"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/pathsafe"
|
||||
)
|
||||
|
||||
const (
|
||||
maxReceiptBytes = 1 << 20
|
||||
maxIndexBytes = 4 << 20
|
||||
maxSummaryBytes = 4 << 20
|
||||
canonicalIndexFile = "index.json"
|
||||
canonicalManifestFile = "manifest.json"
|
||||
canonicalRejectedFile = "rejected.json"
|
||||
canonicalWarningsFile = "warnings.json"
|
||||
maxReceiptBytes = 1 << 20
|
||||
maxIndexBytes = 4 << 20
|
||||
maxSummaryBytes = 4 << 20
|
||||
canonicalIndexFile = "index.json"
|
||||
canonicalManifestFile = "manifest.json"
|
||||
canonicalRejectedFile = "rejected.json"
|
||||
canonicalWarningsFile = "warnings.json"
|
||||
canonicalDiagnosticsFile = "diagnostics.json"
|
||||
warningsSchemaVersion = "notarius.warnings.v2"
|
||||
diagnosticsSchemaVersion = "notarius.diagnostics.v1"
|
||||
maxWarningGroups = 128
|
||||
maxDiagnosticGroups = 256
|
||||
maxFindingSamples = 3
|
||||
)
|
||||
|
||||
type subprocessRun func(context.Context, subprocess.RunRequest) (subprocess.RunResult, error)
|
||||
@@ -41,7 +48,8 @@ func (r *SubprocessRunner) Run(ctx context.Context, req RunRequest) (RunResult,
|
||||
if r == nil || r.run == nil {
|
||||
return RunResult{}, fmt.Errorf("notarius subprocess runner is nil")
|
||||
}
|
||||
if err := validateRunRequest(req); err != nil {
|
||||
references, err := validateRunRequest(req)
|
||||
if err != nil {
|
||||
return RunResult{}, err
|
||||
}
|
||||
|
||||
@@ -50,15 +58,19 @@ func (r *SubprocessRunner) Run(ctx context.Context, req RunRequest) (RunResult,
|
||||
"--config", req.ConfigPath,
|
||||
"--input", req.InputPath,
|
||||
"--output-dir", req.OutputRoot,
|
||||
"--json",
|
||||
}
|
||||
for _, reference := range references {
|
||||
args = append(args, "--reference", reference.Selector+"="+reference.Path)
|
||||
}
|
||||
args = append(args, "--json")
|
||||
processResult, err := r.run(ctx, subprocess.RunRequest{
|
||||
Executable: req.Binary,
|
||||
Args: args,
|
||||
WorkingDir: req.WorkingDirectory,
|
||||
Timeout: req.Timeout,
|
||||
StdoutLogPath: req.ReceiptPath,
|
||||
StderrLogPath: req.LogPath,
|
||||
Executable: req.Binary,
|
||||
Args: args,
|
||||
WorkingDir: req.WorkingDirectory,
|
||||
Timeout: req.Timeout,
|
||||
DiagnosticOwner: "notarius",
|
||||
StdoutLogPath: req.ReceiptPath,
|
||||
StderrLogPath: req.LogPath,
|
||||
})
|
||||
baseResult := RunResult{
|
||||
ReceiptPath: req.ReceiptPath,
|
||||
@@ -94,24 +106,42 @@ func (r *SubprocessRunner) Run(ctx context.Context, req RunRequest) (RunResult,
|
||||
if err != nil {
|
||||
return baseResult, err
|
||||
}
|
||||
diagnostics, diagnosticOccurrences, diagnosticsTruncated, err := loadDiagnostics(index.DiagnosticsPath)
|
||||
if err != nil {
|
||||
return baseResult, err
|
||||
}
|
||||
if receipt.NormalizedOutputCount != len(index.Lanes) || receipt.RejectedOutputCount != len(rejections) ||
|
||||
receipt.WarningGroupCount != len(warnings) || receipt.DiagnosticGroupCount != len(diagnostics) {
|
||||
return baseResult, fmt.Errorf("notarius receipt counts do not match published bundle")
|
||||
}
|
||||
warningOccurrences, err := sumWarningOccurrences(warnings)
|
||||
if err != nil {
|
||||
return baseResult, err
|
||||
}
|
||||
if receipt.WarningOccurrenceCount != warningOccurrences ||
|
||||
receipt.DiagnosticOccurrenceCount != diagnosticOccurrences ||
|
||||
receipt.DiagnosticsTruncated != diagnosticsTruncated {
|
||||
return baseResult, fmt.Errorf("notarius receipt occurrence counts do not match published bundle")
|
||||
}
|
||||
|
||||
baseResult.Receipt = receipt
|
||||
baseResult.Index = index
|
||||
baseResult.BundleRoot = bundleRoot
|
||||
baseResult.Rejections = rejections
|
||||
baseResult.Warnings = warnings
|
||||
baseResult.Diagnostics = diagnostics
|
||||
return baseResult, nil
|
||||
}
|
||||
|
||||
func validateRunRequest(req RunRequest) error {
|
||||
func validateRunRequest(req RunRequest) ([]ReferenceBinding, error) {
|
||||
if strings.TrimSpace(req.Binary) == "" {
|
||||
return fmt.Errorf("notarius binary is required")
|
||||
return nil, fmt.Errorf("notarius binary is required")
|
||||
}
|
||||
if strings.TrimSpace(req.PipelineID) == "" {
|
||||
return fmt.Errorf("notarius pipeline id is required")
|
||||
return nil, fmt.Errorf("notarius pipeline id is required")
|
||||
}
|
||||
if req.Timeout <= 0 {
|
||||
return fmt.Errorf("notarius timeout must be positive")
|
||||
return nil, fmt.Errorf("notarius timeout must be positive")
|
||||
}
|
||||
for label, path := range map[string]string{
|
||||
"config": req.ConfigPath,
|
||||
@@ -122,47 +152,124 @@ func validateRunRequest(req RunRequest) error {
|
||||
"log": req.LogPath,
|
||||
} {
|
||||
if strings.TrimSpace(path) == "" {
|
||||
return fmt.Errorf("notarius %s path is required", label)
|
||||
return nil, fmt.Errorf("notarius %s path is required", label)
|
||||
}
|
||||
if !filepath.IsAbs(path) {
|
||||
return fmt.Errorf("notarius %s path must be absolute", label)
|
||||
return nil, fmt.Errorf("notarius %s path must be absolute", label)
|
||||
}
|
||||
}
|
||||
if filepath.Clean(req.ReceiptPath) == filepath.Clean(req.LogPath) {
|
||||
return fmt.Errorf("notarius receipt and log paths must be different")
|
||||
return nil, fmt.Errorf("notarius receipt and log paths must be different")
|
||||
}
|
||||
references := make([]ReferenceBinding, 0, len(req.References))
|
||||
selectors := make(map[string]struct{}, len(req.References))
|
||||
for index, binding := range req.References {
|
||||
selector, err := notariusref.NormalizeSelector(binding.Selector)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("notarius reference %d selector: %w", index, err)
|
||||
}
|
||||
if _, duplicate := selectors[selector]; duplicate {
|
||||
return nil, fmt.Errorf("notarius reference selector %q is duplicated", selector)
|
||||
}
|
||||
selectors[selector] = struct{}{}
|
||||
if strings.TrimSpace(binding.Path) == "" {
|
||||
return nil, fmt.Errorf("notarius reference %q path is required", selector)
|
||||
}
|
||||
if !filepath.IsAbs(binding.Path) {
|
||||
return nil, fmt.Errorf("notarius reference %q path must be absolute", selector)
|
||||
}
|
||||
references = append(references, ReferenceBinding{Selector: selector, Path: binding.Path})
|
||||
}
|
||||
if err := requireRegularFile(req.ConfigPath); err != nil {
|
||||
return fmt.Errorf("validate notarius config path: %w", err)
|
||||
return nil, fmt.Errorf("validate notarius config path: %w", err)
|
||||
}
|
||||
if err := requireRegularFile(req.InputPath); err != nil {
|
||||
return fmt.Errorf("validate notarius input path: %w", err)
|
||||
return nil, fmt.Errorf("validate notarius input path: %w", err)
|
||||
}
|
||||
if err := requireDirectory(req.OutputRoot); err != nil {
|
||||
return fmt.Errorf("validate notarius output root: %w", err)
|
||||
return nil, fmt.Errorf("validate notarius output root: %w", err)
|
||||
}
|
||||
if err := requireDirectory(req.WorkingDirectory); err != nil {
|
||||
return fmt.Errorf("validate notarius working directory: %w", err)
|
||||
return nil, fmt.Errorf("validate notarius working directory: %w", err)
|
||||
}
|
||||
if err := validateLogDestination(req.ReceiptPath); err != nil {
|
||||
return fmt.Errorf("validate notarius receipt path: %w", err)
|
||||
return nil, fmt.Errorf("validate notarius receipt path: %w", err)
|
||||
}
|
||||
if err := validateLogDestination(req.LogPath); err != nil {
|
||||
return fmt.Errorf("validate notarius log path: %w", err)
|
||||
return nil, fmt.Errorf("validate notarius log path: %w", err)
|
||||
}
|
||||
return nil
|
||||
return references, nil
|
||||
}
|
||||
|
||||
type receiptDocument struct {
|
||||
SchemaVersion string `json:"schema_version"`
|
||||
RunID string `json:"run_id"`
|
||||
PipelineID string `json:"pipeline_id"`
|
||||
OutputDirectory string `json:"output_directory"`
|
||||
IndexFile string `json:"index_file"`
|
||||
NormalizedOutputCount *int `json:"normalized_output_count"`
|
||||
RejectedOutputCount *int `json:"rejected_output_count"`
|
||||
WarningCount *int `json:"warning_count"`
|
||||
ValidationStatus string `json:"validation_status"`
|
||||
DebugDirectory string `json:"debug_directory"`
|
||||
SchemaVersion string `json:"schema_version"`
|
||||
RunID string `json:"run_id"`
|
||||
PipelineID string `json:"pipeline_id"`
|
||||
OutputDirectory string `json:"output_directory"`
|
||||
IndexFile string `json:"index_file"`
|
||||
NormalizedOutputCount *int `json:"normalized_output_count"`
|
||||
RejectedOutputCount *int `json:"rejected_output_count"`
|
||||
WarningGroupCount *int `json:"warning_group_count"`
|
||||
WarningOccurrenceCount *int `json:"warning_occurrence_count"`
|
||||
DiagnosticGroupCount *int `json:"diagnostic_group_count"`
|
||||
DiagnosticOccurrenceCount *int `json:"diagnostic_occurrence_count"`
|
||||
DiagnosticsTruncated *bool `json:"diagnostics_truncated"`
|
||||
ValidationStatus string `json:"validation_status"`
|
||||
ValidationSummaries []validationSummaryDocument `json:"validation_summaries"`
|
||||
DebugDirectory string `json:"debug_directory"`
|
||||
}
|
||||
|
||||
type validationSummaryDocument struct {
|
||||
Stage string `json:"stage"`
|
||||
StepID string `json:"step_id"`
|
||||
LaneID string `json:"lane_id"`
|
||||
ModuleKey string `json:"module_key"`
|
||||
ChunkID string `json:"chunk_id"`
|
||||
Status string `json:"status"`
|
||||
RejectingValidators []string `json:"rejecting_validators"`
|
||||
ReasonCodes []string `json:"reason_codes"`
|
||||
IncompleteValidators []string `json:"incomplete_validators"`
|
||||
ProducerAttemptCount *int `json:"producer_attempt_count"`
|
||||
TerminalAction string `json:"terminal_action"`
|
||||
}
|
||||
|
||||
func validValidationStatus(value string) bool {
|
||||
switch value {
|
||||
case "approved", "rejected", "incomplete":
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
func validateValidationSummaries(documents []validationSummaryDocument) ([]ValidationSummary, error) {
|
||||
summaries := make([]ValidationSummary, 0, len(documents))
|
||||
for _, document := range documents {
|
||||
if document.Status != "complete" && document.Status != "rejected" && document.Status != "incomplete" {
|
||||
return nil, fmt.Errorf("notarius validation summary status %q is invalid", document.Status)
|
||||
}
|
||||
if document.ProducerAttemptCount == nil || *document.ProducerAttemptCount <= 0 || !validTerminalAction(document.TerminalAction) {
|
||||
return nil, fmt.Errorf("notarius validation summary is missing required fields")
|
||||
}
|
||||
summaries = append(summaries, ValidationSummary{
|
||||
Stage: document.Stage, StepID: document.StepID, LaneID: document.LaneID,
|
||||
ModuleKey: document.ModuleKey, ChunkID: document.ChunkID, Status: document.Status,
|
||||
RejectingValidators: append([]string(nil), document.RejectingValidators...),
|
||||
ReasonCodes: append([]string(nil), document.ReasonCodes...),
|
||||
IncompleteValidators: append([]string(nil), document.IncompleteValidators...),
|
||||
ProducerAttemptCount: *document.ProducerAttemptCount, TerminalAction: document.TerminalAction,
|
||||
})
|
||||
}
|
||||
return summaries, nil
|
||||
}
|
||||
|
||||
func validTerminalAction(value string) bool {
|
||||
switch value {
|
||||
case "accepted", "reject_output", "warn_continue", "fail_run":
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
func loadReceipt(path, pipelineID string) (Receipt, error) {
|
||||
@@ -176,7 +283,9 @@ func loadReceipt(path, pipelineID string) (Receipt, error) {
|
||||
if strings.TrimSpace(document.RunID) == "" || strings.TrimSpace(document.PipelineID) == "" ||
|
||||
strings.TrimSpace(document.OutputDirectory) == "" || strings.TrimSpace(document.ValidationStatus) == "" ||
|
||||
document.NormalizedOutputCount == nil ||
|
||||
document.RejectedOutputCount == nil || document.WarningCount == nil {
|
||||
document.RejectedOutputCount == nil || document.WarningGroupCount == nil ||
|
||||
document.WarningOccurrenceCount == nil || document.DiagnosticGroupCount == nil ||
|
||||
document.DiagnosticOccurrenceCount == nil || document.DiagnosticsTruncated == nil {
|
||||
return Receipt{}, fmt.Errorf("notarius receipt is missing required fields")
|
||||
}
|
||||
if document.IndexFile != canonicalIndexFile {
|
||||
@@ -185,9 +294,18 @@ func loadReceipt(path, pipelineID string) (Receipt, error) {
|
||||
if document.PipelineID != pipelineID {
|
||||
return Receipt{}, fmt.Errorf("notarius receipt pipeline id %q does not match requested pipeline %q", document.PipelineID, pipelineID)
|
||||
}
|
||||
if *document.NormalizedOutputCount < 0 || *document.RejectedOutputCount < 0 || *document.WarningCount < 0 {
|
||||
if *document.NormalizedOutputCount < 0 || *document.RejectedOutputCount < 0 ||
|
||||
*document.WarningGroupCount < 0 || *document.WarningOccurrenceCount < 0 ||
|
||||
*document.DiagnosticGroupCount < 0 || *document.DiagnosticOccurrenceCount < 0 {
|
||||
return Receipt{}, fmt.Errorf("notarius receipt counts must be non-negative")
|
||||
}
|
||||
if !validValidationStatus(document.ValidationStatus) {
|
||||
return Receipt{}, fmt.Errorf("notarius receipt validation_status %q is invalid", document.ValidationStatus)
|
||||
}
|
||||
validationSummaries, err := validateValidationSummaries(document.ValidationSummaries)
|
||||
if err != nil {
|
||||
return Receipt{}, err
|
||||
}
|
||||
if !filepath.IsAbs(document.OutputDirectory) {
|
||||
return Receipt{}, fmt.Errorf("notarius receipt output directory must be absolute")
|
||||
}
|
||||
@@ -195,16 +313,21 @@ func loadReceipt(path, pipelineID string) (Receipt, error) {
|
||||
return Receipt{}, fmt.Errorf("notarius receipt debug directory must be absolute when present")
|
||||
}
|
||||
return Receipt{
|
||||
SchemaVersion: document.SchemaVersion,
|
||||
RunID: document.RunID,
|
||||
PipelineID: document.PipelineID,
|
||||
OutputDirectory: filepath.Clean(document.OutputDirectory),
|
||||
IndexFile: document.IndexFile,
|
||||
NormalizedOutputCount: *document.NormalizedOutputCount,
|
||||
RejectedOutputCount: *document.RejectedOutputCount,
|
||||
WarningCount: *document.WarningCount,
|
||||
ValidationStatus: document.ValidationStatus,
|
||||
DebugDirectory: document.DebugDirectory,
|
||||
SchemaVersion: document.SchemaVersion,
|
||||
RunID: document.RunID,
|
||||
PipelineID: document.PipelineID,
|
||||
OutputDirectory: filepath.Clean(document.OutputDirectory),
|
||||
IndexFile: document.IndexFile,
|
||||
NormalizedOutputCount: *document.NormalizedOutputCount,
|
||||
RejectedOutputCount: *document.RejectedOutputCount,
|
||||
WarningGroupCount: *document.WarningGroupCount,
|
||||
WarningOccurrenceCount: *document.WarningOccurrenceCount,
|
||||
DiagnosticGroupCount: *document.DiagnosticGroupCount,
|
||||
DiagnosticOccurrenceCount: *document.DiagnosticOccurrenceCount,
|
||||
DiagnosticsTruncated: *document.DiagnosticsTruncated,
|
||||
ValidationStatus: document.ValidationStatus,
|
||||
ValidationSummaries: validationSummaries,
|
||||
DebugDirectory: document.DebugDirectory,
|
||||
}, nil
|
||||
}
|
||||
|
||||
@@ -213,6 +336,7 @@ type indexDocument struct {
|
||||
OutputFiles *[]laneDocument `json:"output_files"`
|
||||
RejectedFile string `json:"rejected_file"`
|
||||
WarningsFile string `json:"warnings_file"`
|
||||
DiagnosticsFile string `json:"diagnostics_file"`
|
||||
ChunkMap *pipelineDocument `json:"chunk_map"`
|
||||
EvidenceContext *pipelineDocument `json:"evidence_context"`
|
||||
}
|
||||
@@ -249,6 +373,7 @@ func loadIndex(bundleRoot, indexPath string) (Index, error) {
|
||||
{name: "manifest_file", got: document.ManifestFile, want: canonicalManifestFile},
|
||||
{name: "rejected_file", got: document.RejectedFile, want: canonicalRejectedFile},
|
||||
{name: "warnings_file", got: document.WarningsFile, want: canonicalWarningsFile},
|
||||
{name: "diagnostics_file", got: document.DiagnosticsFile, want: canonicalDiagnosticsFile},
|
||||
} {
|
||||
if field.got != field.want {
|
||||
return Index{}, fmt.Errorf("notarius index %s %q is incompatible; want %q", field.name, field.got, field.want)
|
||||
@@ -259,10 +384,11 @@ func loadIndex(bundleRoot, indexPath string) (Index, error) {
|
||||
}
|
||||
|
||||
index := Index{
|
||||
Path: indexPath,
|
||||
ManifestFile: document.ManifestFile,
|
||||
RejectedFile: document.RejectedFile,
|
||||
WarningsFile: document.WarningsFile,
|
||||
Path: indexPath,
|
||||
ManifestFile: document.ManifestFile,
|
||||
RejectedFile: document.RejectedFile,
|
||||
WarningsFile: document.WarningsFile,
|
||||
DiagnosticsFile: document.DiagnosticsFile,
|
||||
}
|
||||
var err error
|
||||
if index.ManifestPath, err = resolveRegularFile(bundleRoot, index.ManifestFile); err != nil {
|
||||
@@ -274,6 +400,9 @@ func loadIndex(bundleRoot, indexPath string) (Index, error) {
|
||||
if index.WarningsPath, err = resolveRegularFile(bundleRoot, index.WarningsFile); err != nil {
|
||||
return Index{}, fmt.Errorf("resolve notarius warning file: %w", err)
|
||||
}
|
||||
if index.DiagnosticsPath, err = resolveRegularFile(bundleRoot, index.DiagnosticsFile); err != nil {
|
||||
return Index{}, fmt.Errorf("resolve notarius diagnostics file: %w", err)
|
||||
}
|
||||
|
||||
seenLanes := make(map[string]struct{}, len(*document.OutputFiles))
|
||||
for _, lane := range *document.OutputFiles {
|
||||
@@ -361,12 +490,34 @@ func loadRejections(path string) ([]RejectionSummary, error) {
|
||||
return summaries, nil
|
||||
}
|
||||
|
||||
type warningDocument struct {
|
||||
Warnings *[]struct {
|
||||
type findingGroupDocument struct {
|
||||
Disposition string `json:"disposition"`
|
||||
Category string `json:"category"`
|
||||
ReasonCode string `json:"reason_code"`
|
||||
Origin diagnosticOriginDocument `json:"origin"`
|
||||
OccurrenceCount *int `json:"occurrence_count"`
|
||||
Samples *[]struct {
|
||||
Scope string `json:"scope"`
|
||||
ReasonCode string `json:"reason_code"`
|
||||
Message string `json:"message"`
|
||||
} `json:"warnings"`
|
||||
ChunkID string `json:"chunk_id"`
|
||||
ChunkIndex *int `json:"chunk_index"`
|
||||
} `json:"samples"`
|
||||
OmittedSampleCount *int `json:"omitted_sample_count"`
|
||||
}
|
||||
|
||||
type diagnosticOriginDocument struct {
|
||||
Stage string `json:"stage"`
|
||||
StepID string `json:"step_id"`
|
||||
LaneID string `json:"lane_id"`
|
||||
ModuleKey string `json:"module_key"`
|
||||
ValidatorKey string `json:"validator_key"`
|
||||
}
|
||||
|
||||
type warningDocument struct {
|
||||
SchemaVersion string `json:"schema_version"`
|
||||
GroupCount *int `json:"group_count"`
|
||||
OccurrenceCount *int `json:"occurrence_count"`
|
||||
Groups *[]findingGroupDocument `json:"groups"`
|
||||
}
|
||||
|
||||
func loadWarnings(path string) ([]WarningSummary, error) {
|
||||
@@ -374,46 +525,155 @@ func loadWarnings(path string) ([]WarningSummary, error) {
|
||||
if err := decodeBoundedJSON(path, maxSummaryBytes, &document); err != nil {
|
||||
return nil, fmt.Errorf("decode notarius warnings: %w", err)
|
||||
}
|
||||
if document.Warnings == nil {
|
||||
return nil, fmt.Errorf("notarius warning document is missing warnings array")
|
||||
if document.SchemaVersion != warningsSchemaVersion || document.GroupCount == nil ||
|
||||
document.OccurrenceCount == nil || document.Groups == nil {
|
||||
return nil, fmt.Errorf("notarius warning document is missing or incompatible required fields")
|
||||
}
|
||||
summaries := make([]WarningSummary, 0, len(*document.Warnings))
|
||||
for _, item := range *document.Warnings {
|
||||
if strings.TrimSpace(item.ReasonCode) == "" || strings.TrimSpace(item.Message) == "" {
|
||||
return nil, fmt.Errorf("notarius warning entries require reason_code and message")
|
||||
if *document.GroupCount < 0 || *document.GroupCount > maxWarningGroups || *document.OccurrenceCount < 0 ||
|
||||
*document.GroupCount != len(*document.Groups) {
|
||||
return nil, fmt.Errorf("notarius warning document counts are inconsistent")
|
||||
}
|
||||
summaries := make([]WarningSummary, 0, len(*document.Groups))
|
||||
occurrences := 0
|
||||
for _, group := range *document.Groups {
|
||||
if err := validateFindingGroup(group); err != nil {
|
||||
return nil, fmt.Errorf("notarius warning group: %w", err)
|
||||
}
|
||||
summaries = append(summaries, WarningSummary{Scope: item.Scope, ReasonCode: item.ReasonCode})
|
||||
if group.Disposition != "warning" {
|
||||
return nil, fmt.Errorf("notarius warning group disposition %q is invalid", group.Disposition)
|
||||
}
|
||||
if *group.OccurrenceCount > int(^uint(0)>>1)-occurrences {
|
||||
return nil, fmt.Errorf("notarius warning occurrence count overflows")
|
||||
}
|
||||
occurrences += *group.OccurrenceCount
|
||||
summaries = append(summaries, WarningSummary{
|
||||
Disposition: group.Disposition, Category: group.Category, ReasonCode: group.ReasonCode,
|
||||
Origin: diagnosticOrigin(group.Origin), OccurrenceCount: *group.OccurrenceCount,
|
||||
})
|
||||
}
|
||||
if occurrences != *document.OccurrenceCount {
|
||||
return nil, fmt.Errorf("notarius warning document occurrence count is inconsistent")
|
||||
}
|
||||
return summaries, nil
|
||||
}
|
||||
|
||||
type diagnosticDocument struct {
|
||||
SchemaVersion string `json:"schema_version"`
|
||||
GroupCount *int `json:"group_count"`
|
||||
OccurrenceCount *int `json:"occurrence_count"`
|
||||
Truncated *bool `json:"truncated"`
|
||||
UnrepresentedOccurrenceCount *int `json:"unrepresented_occurrence_count"`
|
||||
Groups *[]findingGroupDocument `json:"groups"`
|
||||
}
|
||||
|
||||
func loadDiagnostics(path string) ([]DiagnosticSummary, int, bool, error) {
|
||||
var document diagnosticDocument
|
||||
if err := decodeBoundedJSON(path, maxSummaryBytes, &document); err != nil {
|
||||
return nil, 0, false, fmt.Errorf("decode notarius diagnostics: %w", err)
|
||||
}
|
||||
if document.SchemaVersion != diagnosticsSchemaVersion || document.GroupCount == nil ||
|
||||
document.OccurrenceCount == nil || document.Truncated == nil ||
|
||||
document.UnrepresentedOccurrenceCount == nil || document.Groups == nil {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostics document is missing or incompatible required fields")
|
||||
}
|
||||
if *document.GroupCount < 0 || *document.GroupCount > maxDiagnosticGroups || *document.OccurrenceCount < 0 ||
|
||||
*document.UnrepresentedOccurrenceCount < 0 || *document.GroupCount != len(*document.Groups) {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostics document counts are inconsistent")
|
||||
}
|
||||
if !*document.Truncated && *document.UnrepresentedOccurrenceCount != 0 {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostics document has unrepresented occurrences without truncation")
|
||||
}
|
||||
summaries := make([]DiagnosticSummary, 0, len(*document.Groups))
|
||||
representedOccurrences := 0
|
||||
for _, group := range *document.Groups {
|
||||
if err := validateFindingGroup(group); err != nil {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostic group: %w", err)
|
||||
}
|
||||
if group.Disposition != "advisory" && group.Disposition != "observation" {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostic group disposition %q is invalid", group.Disposition)
|
||||
}
|
||||
if *group.OccurrenceCount > int(^uint(0)>>1)-representedOccurrences {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostic occurrence count overflows")
|
||||
}
|
||||
representedOccurrences += *group.OccurrenceCount
|
||||
summaries = append(summaries, DiagnosticSummary{
|
||||
Disposition: group.Disposition, Category: group.Category, ReasonCode: group.ReasonCode,
|
||||
Origin: diagnosticOrigin(group.Origin), OccurrenceCount: *group.OccurrenceCount,
|
||||
})
|
||||
}
|
||||
if *document.UnrepresentedOccurrenceCount > int(^uint(0)>>1)-representedOccurrences ||
|
||||
representedOccurrences+*document.UnrepresentedOccurrenceCount != *document.OccurrenceCount {
|
||||
return nil, 0, false, fmt.Errorf("notarius diagnostics document occurrence count is inconsistent")
|
||||
}
|
||||
return summaries, *document.OccurrenceCount, *document.Truncated, nil
|
||||
}
|
||||
|
||||
func validateFindingGroup(group findingGroupDocument) error {
|
||||
if strings.TrimSpace(group.Disposition) == "" || strings.TrimSpace(group.Category) == "" ||
|
||||
strings.TrimSpace(group.ReasonCode) == "" || !validDiagnosticOriginStage(group.Origin.Stage) ||
|
||||
!validDiagnosticCategory(group.Disposition, group.Category) ||
|
||||
group.OccurrenceCount == nil || *group.OccurrenceCount <= 0 || group.Samples == nil ||
|
||||
group.OmittedSampleCount == nil || *group.OmittedSampleCount < 0 {
|
||||
return fmt.Errorf("missing required fields")
|
||||
}
|
||||
if len(*group.Samples) == 0 || len(*group.Samples) > maxFindingSamples ||
|
||||
*group.OmittedSampleCount != *group.OccurrenceCount-len(*group.Samples) {
|
||||
return fmt.Errorf("sample counts are inconsistent")
|
||||
}
|
||||
for _, sample := range *group.Samples {
|
||||
if strings.TrimSpace(sample.Scope) == "" || strings.TrimSpace(sample.Message) == "" ||
|
||||
(sample.ChunkIndex != nil && *sample.ChunkIndex < 0) {
|
||||
return fmt.Errorf("samples require scope and message")
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func validDiagnosticCategory(disposition, category string) bool {
|
||||
switch disposition {
|
||||
case "warning":
|
||||
return category == "configuration" || category == "degradation" ||
|
||||
category == "validation_incomplete" || category == "fallback"
|
||||
case "advisory":
|
||||
return category == "data_quality"
|
||||
case "observation":
|
||||
return category == "normalization"
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
func validDiagnosticOriginStage(stage string) bool {
|
||||
switch stage {
|
||||
case "references", "chunk", "extract", "merge", "normalize":
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
func diagnosticOrigin(document diagnosticOriginDocument) DiagnosticOrigin {
|
||||
return DiagnosticOrigin{
|
||||
Stage: document.Stage, StepID: document.StepID, LaneID: document.LaneID,
|
||||
ModuleKey: document.ModuleKey, ValidatorKey: document.ValidatorKey,
|
||||
}
|
||||
}
|
||||
|
||||
func sumWarningOccurrences(values []WarningSummary) (int, error) {
|
||||
total := 0
|
||||
for _, value := range values {
|
||||
if value.OccurrenceCount > int(^uint(0)>>1)-total {
|
||||
return 0, fmt.Errorf("notarius warning occurrence count overflows")
|
||||
}
|
||||
total += value.OccurrenceCount
|
||||
}
|
||||
return total, nil
|
||||
}
|
||||
|
||||
func decodeBoundedJSON(path string, limit int64, destination any) error {
|
||||
inspected, err := os.Lstat(path)
|
||||
data, err := fileops.ReadRegularFile(path, limit)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if inspected.Mode()&os.ModeSymlink != 0 || !inspected.Mode().IsRegular() {
|
||||
return fmt.Errorf("path %q must be a regular file without symlinks", path)
|
||||
}
|
||||
file, err := os.Open(path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer func() { _ = file.Close() }()
|
||||
opened, err := file.Stat()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if !opened.Mode().IsRegular() || !os.SameFile(inspected, opened) {
|
||||
return fmt.Errorf("file %q changed before it could be read", path)
|
||||
}
|
||||
reader := io.LimitReader(file, limit+1)
|
||||
data, err := io.ReadAll(reader)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if int64(len(data)) > limit {
|
||||
return fmt.Errorf("file %q exceeds %d-byte limit", path, limit)
|
||||
return fmt.Errorf("notarius JSON result exceeds or cannot be read within %d-byte limit: %w", limit, err)
|
||||
}
|
||||
if err := json.Unmarshal(data, destination); err != nil {
|
||||
return err
|
||||
|
||||
@@ -52,6 +52,10 @@ func TestSubprocessRunnerBuildsExactInvocationAndDiscoversBundle(t *testing.T) {
|
||||
if result.Receipt.SchemaVersion != ReceiptSchemaVersion || result.Receipt.RunID != "notarius-run-1" {
|
||||
t.Fatalf("receipt = %#v", result.Receipt)
|
||||
}
|
||||
if len(result.Receipt.ValidationSummaries) != 1 || result.Receipt.ValidationSummaries[0].LaneID != "npc-registry" ||
|
||||
result.Receipt.ValidationSummaries[0].Status != "complete" {
|
||||
t.Fatalf("validation summaries = %#v", result.Receipt.ValidationSummaries)
|
||||
}
|
||||
if len(result.Index.Lanes) != 1 || result.Index.Lanes[0].LaneID != "npc-registry" {
|
||||
t.Fatalf("lanes = %#v", result.Index.Lanes)
|
||||
}
|
||||
@@ -64,12 +68,110 @@ func TestSubprocessRunnerBuildsExactInvocationAndDiscoversBundle(t *testing.T) {
|
||||
if len(result.Rejections) != 1 || result.Rejections[0].LaneID != "spells" || result.Rejections[0].ReasonCode != "invalid_spell" {
|
||||
t.Fatalf("rejections = %#v", result.Rejections)
|
||||
}
|
||||
if len(result.Warnings) != 1 || result.Warnings[0].Scope != "lane:npc-registry" || result.Warnings[0].ReasonCode != "normalized_name" {
|
||||
if len(result.Warnings) != 1 || result.Warnings[0].Category != "degradation" || result.Warnings[0].ReasonCode != "normalized_name" {
|
||||
t.Fatalf("warnings = %#v", result.Warnings)
|
||||
}
|
||||
if len(result.Diagnostics) != 1 || result.Diagnostics[0].Category != "data_quality" || result.Diagnostics[0].ReasonCode != "low_confidence" {
|
||||
t.Fatalf("diagnostics = %#v", result.Diagnostics)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSubprocessRunnerInheritsEnvironmentAndSeparatesStreams(t *testing.T) {
|
||||
func TestSubprocessRunnerBuildsOrderedReferenceArguments(t *testing.T) {
|
||||
req := validRunRequest(t)
|
||||
referenceRoot := t.TempDir()
|
||||
req.References = []ReferenceBinding{
|
||||
{Selector: " party ", Path: filepath.Join(referenceRoot, "party context=primary.json")},
|
||||
{Selector: " npc-registry . extract . glossary ", Path: filepath.Join(referenceRoot, "glossary.json")},
|
||||
}
|
||||
originalReferences := append([]ReferenceBinding(nil), req.References...)
|
||||
var captured sharedsubprocess.RunRequest
|
||||
runner := &SubprocessRunner{run: func(_ context.Context, processReq sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
|
||||
captured = processReq
|
||||
writeValidBundleAndReceipt(t, req, false)
|
||||
return sharedsubprocess.RunResult{ExitCode: 0}, nil
|
||||
}}
|
||||
|
||||
if _, err := runner.Run(context.Background(), req); err != nil {
|
||||
t.Fatalf("Run() error = %v", err)
|
||||
}
|
||||
wantArgs := []string{
|
||||
"run", req.PipelineID,
|
||||
"--config", req.ConfigPath,
|
||||
"--input", req.InputPath,
|
||||
"--output-dir", req.OutputRoot,
|
||||
"--reference", "party=" + req.References[0].Path,
|
||||
"--reference", "npc-registry.extract.glossary=" + req.References[1].Path,
|
||||
"--json",
|
||||
}
|
||||
if !reflect.DeepEqual(captured.Args, wantArgs) {
|
||||
t.Fatalf("subprocess args = %#v, want %#v", captured.Args, wantArgs)
|
||||
}
|
||||
if !reflect.DeepEqual(req.References, originalReferences) {
|
||||
t.Fatalf("Run() mutated caller references = %#v, want %#v", req.References, originalReferences)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSubprocessRunnerRejectsInvalidReferencesBeforeLaunch(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
references func(string) []ReferenceBinding
|
||||
wantErr string
|
||||
}{
|
||||
{
|
||||
name: "invalid selector",
|
||||
references: func(root string) []ReferenceBinding {
|
||||
return []ReferenceBinding{{Selector: "lane.prepare.party", Path: filepath.Join(root, "party.json")}}
|
||||
},
|
||||
wantErr: "selector",
|
||||
},
|
||||
{
|
||||
name: "duplicate normalized selector",
|
||||
references: func(root string) []ReferenceBinding {
|
||||
return []ReferenceBinding{
|
||||
{Selector: "lane.party", Path: filepath.Join(root, "party.json")},
|
||||
{Selector: " lane . party ", Path: filepath.Join(root, "party-2.json")},
|
||||
}
|
||||
},
|
||||
wantErr: "duplicated",
|
||||
},
|
||||
{
|
||||
name: "empty path",
|
||||
references: func(string) []ReferenceBinding {
|
||||
return []ReferenceBinding{{Selector: "party", Path: " "}}
|
||||
},
|
||||
wantErr: "path is required",
|
||||
},
|
||||
{
|
||||
name: "relative path",
|
||||
references: func(string) []ReferenceBinding {
|
||||
return []ReferenceBinding{{Selector: "party", Path: "references/party.json"}}
|
||||
},
|
||||
wantErr: "path must be absolute",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
req := validRunRequest(t)
|
||||
req.References = tt.references(t.TempDir())
|
||||
started := false
|
||||
runner := &SubprocessRunner{run: func(context.Context, sharedsubprocess.RunRequest) (sharedsubprocess.RunResult, error) {
|
||||
started = true
|
||||
return sharedsubprocess.RunResult{}, nil
|
||||
}}
|
||||
|
||||
_, err := runner.Run(context.Background(), req)
|
||||
if err == nil || !strings.Contains(err.Error(), tt.wantErr) {
|
||||
t.Fatalf("Run() error = %v, want containing %q", err, tt.wantErr)
|
||||
}
|
||||
if started {
|
||||
t.Fatal("subprocess started after request validation failure")
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestSubprocessRunnerUsesMinimalEnvironmentAndSeparatesStreams(t *testing.T) {
|
||||
req := validRunRequest(t)
|
||||
writeValidBundleAndReceipt(t, req, false)
|
||||
receiptFixture := req.ReceiptPath + ".fixture"
|
||||
@@ -103,7 +205,7 @@ cat "$NOTARIUS_RECEIPT_FIXTURE"
|
||||
t.Fatalf("Run() error = %v", err)
|
||||
}
|
||||
assertTextFile(t, filepath.Join(captureDir, "working-directory"), req.WorkingDirectory+"\n")
|
||||
assertTextFile(t, filepath.Join(captureDir, "environment"), "inherited-value")
|
||||
assertTextFile(t, filepath.Join(captureDir, "environment"), "")
|
||||
assertTextFile(t, req.LogPath, "diagnostic stream\n")
|
||||
receiptBytes, err := os.ReadFile(req.ReceiptPath)
|
||||
if err != nil {
|
||||
@@ -171,8 +273,10 @@ func TestLoadReceiptValidation(t *testing.T) {
|
||||
valid := map[string]any{
|
||||
"schema_version": ReceiptSchemaVersion, "run_id": "run-1", "pipeline_id": "pipeline-1",
|
||||
"output_directory": filepath.Join(root, "outputs", "run-1"), "index_file": "index.json",
|
||||
"normalized_output_count": 1, "rejected_output_count": 0, "warning_count": 0,
|
||||
"validation_status": "approved", "future_field": true,
|
||||
"normalized_output_count": 1, "rejected_output_count": 0,
|
||||
"warning_group_count": 0, "warning_occurrence_count": 0,
|
||||
"diagnostic_group_count": 0, "diagnostic_occurrence_count": 0,
|
||||
"diagnostics_truncated": false, "validation_status": "approved", "future_field": true,
|
||||
}
|
||||
tests := []struct {
|
||||
name string
|
||||
@@ -183,11 +287,12 @@ func TestLoadReceiptValidation(t *testing.T) {
|
||||
}{
|
||||
{name: "unknown fields tolerated", wantOK: true},
|
||||
{name: "malformed", raw: []byte("{")},
|
||||
{name: "unsupported version", mutate: func(v map[string]any) { v["schema_version"] = "notarius.run-result.v2" }},
|
||||
{name: "unsupported version", mutate: func(v map[string]any) { v["schema_version"] = "notarius.run-result.v1" }},
|
||||
{name: "missing field", mutate: func(v map[string]any) { delete(v, "run_id") }},
|
||||
{name: "pipeline mismatch", mutate: func(v map[string]any) { v["pipeline_id"] = "other" }},
|
||||
{name: "relative output", mutate: func(v map[string]any) { v["output_directory"] = "run-1" }},
|
||||
{name: "negative count", mutate: func(v map[string]any) { v["warning_count"] = -1 }},
|
||||
{name: "negative count", mutate: func(v map[string]any) { v["warning_group_count"] = -1 }},
|
||||
{name: "invalid validation status", mutate: func(v map[string]any) { v["validation_status"] = "valid" }},
|
||||
{
|
||||
name: "nested index", mutate: func(v map[string]any) { v["index_file"] = "nested/index.json" },
|
||||
wantError: `index_file "nested/index.json"`,
|
||||
@@ -369,22 +474,26 @@ func TestLoadDiagnosticSummariesValidateBoundsAndTolerateUnknownFields(t *testin
|
||||
root := t.TempDir()
|
||||
rejectedPath := filepath.Join(root, "rejected.json")
|
||||
warningsPath := filepath.Join(root, "warnings.json")
|
||||
diagnosticsPath := filepath.Join(root, "diagnostics.json")
|
||||
writeJSONFile(t, rejectedPath, map[string]any{"rejected": []any{map[string]any{
|
||||
"stage": "validate", "lane_id": "spells", "reason_code": "invalid", "message": "do not retain this", "future": true,
|
||||
}}, "future": true})
|
||||
writeJSONFile(t, warningsPath, map[string]any{"warnings": []any{map[string]any{
|
||||
"scope": "lane:spells", "reason_code": "bounded", "message": "do not retain this", "future": true,
|
||||
}}, "future": true})
|
||||
writeJSONFile(t, warningsPath, findingEnvelope(warningsSchemaVersion, []any{findingGroup("warning", "degradation", "bounded", "normalize", 2)}, 2, false, 0))
|
||||
writeJSONFile(t, diagnosticsPath, findingEnvelope(diagnosticsSchemaVersion, []any{findingGroup("advisory", "data_quality", "low_confidence", "normalize", 3)}, 4, true, 1))
|
||||
rejections, err := loadRejections(rejectedPath)
|
||||
if err != nil || len(rejections) != 1 || rejections[0].ReasonCode != "invalid" {
|
||||
t.Fatalf("loadRejections() = %#v, %v", rejections, err)
|
||||
}
|
||||
warnings, err := loadWarnings(warningsPath)
|
||||
if err != nil || len(warnings) != 1 || warnings[0].Scope != "lane:spells" {
|
||||
if err != nil || len(warnings) != 1 || warnings[0].Category != "degradation" || warnings[0].OccurrenceCount != 2 {
|
||||
t.Fatalf("loadWarnings() = %#v, %v", warnings, err)
|
||||
}
|
||||
diagnostics, occurrences, truncated, err := loadDiagnostics(diagnosticsPath)
|
||||
if err != nil || len(diagnostics) != 1 || occurrences != 4 || !truncated || diagnostics[0].Category != "data_quality" {
|
||||
t.Fatalf("loadDiagnostics() = %#v, %d, %t, %v", diagnostics, occurrences, truncated, err)
|
||||
}
|
||||
|
||||
for name, path := range map[string]string{"rejections": rejectedPath, "warnings": warningsPath} {
|
||||
for name, path := range map[string]string{"rejections": rejectedPath, "warnings": warningsPath, "diagnostics": diagnosticsPath} {
|
||||
t.Run("malformed "+name, func(t *testing.T) {
|
||||
if err := os.WriteFile(path, []byte("{"), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile() error = %v", err)
|
||||
@@ -392,8 +501,10 @@ func TestLoadDiagnosticSummariesValidateBoundsAndTolerateUnknownFields(t *testin
|
||||
var err error
|
||||
if name == "rejections" {
|
||||
_, err = loadRejections(path)
|
||||
} else {
|
||||
} else if name == "warnings" {
|
||||
_, err = loadWarnings(path)
|
||||
} else {
|
||||
_, _, _, err = loadDiagnostics(path)
|
||||
}
|
||||
if err == nil {
|
||||
t.Fatal("summary decoder error = nil")
|
||||
@@ -410,16 +521,23 @@ func TestLoadDiagnosticSummariesValidateBoundsAndTolerateUnknownFields(t *testin
|
||||
if _, err := loadRejections(oversized); err == nil || !strings.Contains(err.Error(), "exceeds") {
|
||||
t.Fatalf("loadRejections(oversized) error = %v", err)
|
||||
}
|
||||
if _, _, _, err := loadDiagnostics(oversized); err == nil || !strings.Contains(err.Error(), "exceeds") {
|
||||
t.Fatalf("loadDiagnostics(oversized) error = %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFakeRunnerCapturesRequestsAndHonorsContextAndError(t *testing.T) {
|
||||
req := RunRequest{PipelineID: "pipeline"}
|
||||
req := RunRequest{PipelineID: "pipeline", References: []ReferenceBinding{{Selector: "party", Path: "/references/party.json"}}}
|
||||
want := RunResult{BundleRoot: "/bundle"}
|
||||
fake := &FakeRunner{Result: want}
|
||||
got, err := fake.Run(context.Background(), req)
|
||||
if err != nil || !reflect.DeepEqual(got, want) || !reflect.DeepEqual(fake.Requests, []RunRequest{req}) {
|
||||
t.Fatalf("Run() = %#v, %v; requests = %#v", got, err, fake.Requests)
|
||||
}
|
||||
req.References[0].Path = "/references/changed.json"
|
||||
if fake.Requests[0].References[0].Path != "/references/party.json" {
|
||||
t.Fatalf("fake retained aliased request references: %#v", fake.Requests[0].References)
|
||||
}
|
||||
|
||||
wantErr := errors.New("configured failure")
|
||||
fake.Err = wantErr
|
||||
@@ -478,13 +596,12 @@ func writeValidBundleAndReceipt(t *testing.T, req RunRequest, includeUnknown boo
|
||||
}
|
||||
}
|
||||
rejection := map[string]any{"stage": "validate", "lane_id": "spells", "reason_code": "invalid_spell", "message": strings.Repeat("external detail", 20)}
|
||||
warning := map[string]any{"scope": "lane:npc-registry", "reason_code": "normalized_name", "message": strings.Repeat("external warning", 20)}
|
||||
if includeUnknown {
|
||||
rejection["future"] = true
|
||||
warning["future"] = true
|
||||
}
|
||||
writeJSONFile(t, filepath.Join(bundle, "rejected.json"), map[string]any{"rejected": []any{rejection}, "future": true})
|
||||
writeJSONFile(t, filepath.Join(bundle, "warnings.json"), map[string]any{"warnings": []any{warning}, "future": true})
|
||||
writeJSONFile(t, filepath.Join(bundle, "warnings.json"), findingEnvelope(warningsSchemaVersion, []any{findingGroup("warning", "degradation", "normalized_name", "normalize", 2)}, 2, false, 0))
|
||||
writeJSONFile(t, filepath.Join(bundle, "diagnostics.json"), findingEnvelope(diagnosticsSchemaVersion, []any{findingGroup("advisory", "data_quality", "low_confidence", "normalize", 3)}, 4, true, 1))
|
||||
index := validIndexValue([]any{map[string]any{
|
||||
"lane_id": "npc-registry", "file": "lanes/npc.json", "media_type": "application/json",
|
||||
"module_key": "dnd/npc-registry", "schema_id": "notarius.dnd.npc_registry",
|
||||
@@ -503,7 +620,13 @@ func writeValidBundleAndReceipt(t *testing.T, req RunRequest, includeUnknown boo
|
||||
receipt := map[string]any{
|
||||
"schema_version": ReceiptSchemaVersion, "run_id": "notarius-run-1", "pipeline_id": req.PipelineID,
|
||||
"output_directory": bundle, "index_file": "index.json", "normalized_output_count": 1,
|
||||
"rejected_output_count": 1, "warning_count": 1, "validation_status": "rejected",
|
||||
"rejected_output_count": 1, "warning_group_count": 1, "warning_occurrence_count": 2,
|
||||
"diagnostic_group_count": 1, "diagnostic_occurrence_count": 4,
|
||||
"diagnostics_truncated": true, "validation_status": "rejected",
|
||||
"validation_summaries": []any{map[string]any{
|
||||
"stage": "normalize", "lane_id": "npc-registry", "status": "complete",
|
||||
"producer_attempt_count": 1, "terminal_action": "accepted",
|
||||
}},
|
||||
}
|
||||
if includeUnknown {
|
||||
receipt["future"] = true
|
||||
@@ -517,7 +640,7 @@ func createBundleSkeleton(t *testing.T) string {
|
||||
if err := os.MkdirAll(filepath.Join(bundle, "lanes"), 0o755); err != nil {
|
||||
t.Fatalf("MkdirAll(bundle) error = %v", err)
|
||||
}
|
||||
for _, name := range []string{"manifest.json", "rejected.json", "warnings.json", "lanes/npc.json", "chunk-map.json"} {
|
||||
for _, name := range []string{"manifest.json", "rejected.json", "warnings.json", "diagnostics.json", "lanes/npc.json", "chunk-map.json"} {
|
||||
if err := os.WriteFile(filepath.Join(bundle, filepath.FromSlash(name)), []byte("{}"), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile(%q) error = %v", name, err)
|
||||
}
|
||||
@@ -528,10 +651,31 @@ func createBundleSkeleton(t *testing.T) string {
|
||||
func validIndexValue(lanes []any) map[string]any {
|
||||
return map[string]any{
|
||||
"manifest_file": "manifest.json", "output_files": lanes,
|
||||
"rejected_file": "rejected.json", "warnings_file": "warnings.json",
|
||||
"rejected_file": "rejected.json", "warnings_file": "warnings.json", "diagnostics_file": "diagnostics.json",
|
||||
}
|
||||
}
|
||||
|
||||
func findingGroup(disposition, category, reasonCode, origin string, occurrences int) map[string]any {
|
||||
return map[string]any{
|
||||
"disposition": disposition, "category": category, "reason_code": reasonCode,
|
||||
"origin": map[string]any{"stage": origin, "lane_id": "npc-registry"}, "occurrence_count": occurrences,
|
||||
"samples": []any{map[string]any{"scope": "lane:npc-registry", "message": "external detail"}},
|
||||
"omitted_sample_count": occurrences - 1,
|
||||
}
|
||||
}
|
||||
|
||||
func findingEnvelope(schema string, groups []any, occurrences int, truncated bool, unrepresented int) map[string]any {
|
||||
value := map[string]any{
|
||||
"schema_version": schema, "group_count": len(groups), "occurrence_count": occurrences,
|
||||
"groups": groups,
|
||||
}
|
||||
if schema == diagnosticsSchemaVersion {
|
||||
value["truncated"] = truncated
|
||||
value["unrepresented_occurrence_count"] = unrepresented
|
||||
}
|
||||
return value
|
||||
}
|
||||
|
||||
func writeJSONFile(t *testing.T, path string, value any) {
|
||||
t.Helper()
|
||||
data, err := json.Marshal(value)
|
||||
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"fmt"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
// NoopRunner is a deterministic no-op scriptorium adapter.
|
||||
@@ -142,7 +143,7 @@ func (f *FakeRunner) RenderArtifact(ctx context.Context, req RenderArtifactReque
|
||||
|
||||
func materializeRunPlaceholders(req RunArtifactRequest) error {
|
||||
if req.OutputPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("scriptorium noop/fake run artifact\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("scriptorium noop/fake run artifact\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write run output %q: %w", req.OutputPath, err)
|
||||
}
|
||||
}
|
||||
@@ -154,17 +155,17 @@ func materializeRunPlaceholders(req RunArtifactRequest) error {
|
||||
"prompt_id": req.PromptID,
|
||||
"output_path": req.OutputPath,
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake run stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake run stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake run stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake run stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
@@ -173,7 +174,7 @@ func materializeRunPlaceholders(req RunArtifactRequest) error {
|
||||
|
||||
func materializeRenderPlaceholders(req RenderArtifactRequest) error {
|
||||
if req.OutputPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("{\"schema\":\"scriptorium.render.v1\",\"placeholder\":true}\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputPath, []byte("{\"schema\":\"scriptorium.render.v1\",\"placeholder\":true}\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write render output %q: %w", req.OutputPath, err)
|
||||
}
|
||||
}
|
||||
@@ -185,17 +186,17 @@ func materializeRenderPlaceholders(req RenderArtifactRequest) error {
|
||||
"prompt_id": req.PromptID,
|
||||
"output_path": req.OutputPath,
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake render stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("scriptorium noop/fake render stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake render stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("scriptorium noop/fake render stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,8 +9,12 @@ import (
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
// MaxOutputFileBytes bounds one Scriptorium artifact result.
|
||||
const MaxOutputFileBytes int64 = 64 * 1024 * 1024
|
||||
|
||||
// SubprocessRunner invokes Scriptorium through its public CLI.
|
||||
type SubprocessRunner struct{}
|
||||
|
||||
@@ -52,13 +56,17 @@ func (r *SubprocessRunner) RunArtifact(ctx context.Context, req RunArtifactReque
|
||||
}
|
||||
}
|
||||
|
||||
envOverrides, sensitiveNames := credentialEnvironment(req.APIKeyEnv)
|
||||
runRes, runErr := subprocess.Run(ctx, subprocess.RunRequest{
|
||||
Executable: req.Binary,
|
||||
Args: args,
|
||||
WorkingDir: req.WorkingDir,
|
||||
Timeout: req.Timeout,
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
Executable: req.Binary,
|
||||
Args: args,
|
||||
WorkingDir: req.WorkingDir,
|
||||
Timeout: req.Timeout,
|
||||
EnvOverrides: envOverrides,
|
||||
SensitiveEnvNames: sensitiveNames,
|
||||
DiagnosticOwner: "scriptorium",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
|
||||
result := ArtifactResult{
|
||||
@@ -134,13 +142,17 @@ func (r *SubprocessRunner) RenderArtifact(ctx context.Context, req RenderArtifac
|
||||
}
|
||||
}
|
||||
|
||||
envOverrides, sensitiveNames := credentialEnvironment(req.APIKeyEnv)
|
||||
runRes, runErr := subprocess.Run(ctx, subprocess.RunRequest{
|
||||
Executable: req.Binary,
|
||||
Args: args,
|
||||
WorkingDir: req.WorkingDir,
|
||||
Timeout: req.Timeout,
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
Executable: req.Binary,
|
||||
Args: args,
|
||||
WorkingDir: req.WorkingDir,
|
||||
Timeout: req.Timeout,
|
||||
EnvOverrides: envOverrides,
|
||||
SensitiveEnvNames: sensitiveNames,
|
||||
DiagnosticOwner: "scriptorium",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
|
||||
result := ArtifactResult{
|
||||
@@ -223,6 +235,15 @@ func validateCommonRunRequest(
|
||||
return true, nil
|
||||
}
|
||||
|
||||
func credentialEnvironment(apiKeyEnv string) (map[string]string, []string) {
|
||||
name := strings.TrimSpace(apiKeyEnv)
|
||||
if name == "" {
|
||||
return nil, nil
|
||||
}
|
||||
value, _ := os.LookupEnv(name)
|
||||
return map[string]string{name: value}, []string{name}
|
||||
}
|
||||
|
||||
func buildRunArgs(req RunArtifactRequest) []string {
|
||||
args := []string{"run", "--prompt", strings.TrimSpace(req.PromptID)}
|
||||
if cfgPath := strings.TrimSpace(req.ConfigPath); cfgPath != "" {
|
||||
@@ -321,18 +342,15 @@ func writeInvocationConfig(path string, payload invocationPayload) error {
|
||||
"render_format": payload.RenderFormat,
|
||||
"render_prompt_logged": payload.RenderPromptStore,
|
||||
}
|
||||
return subprocess.WriteYAMLAtomic(path, data, 0o644)
|
||||
return subprocess.WriteYAMLAtomic(path, data, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
func validateNonEmptyOutput(path string) error {
|
||||
info, err := os.Stat(path)
|
||||
data, err := fileops.ReadRegularFile(path, MaxOutputFileBytes)
|
||||
if err != nil {
|
||||
return fmt.Errorf("stat file: %w", err)
|
||||
return fmt.Errorf("scriptorium artifact output exceeds or cannot be read within %d-byte limit: %w", MaxOutputFileBytes, err)
|
||||
}
|
||||
if info.IsDir() {
|
||||
return fmt.Errorf("path is a directory")
|
||||
}
|
||||
if info.Size() <= 0 {
|
||||
if len(data) == 0 {
|
||||
return fmt.Errorf("file is empty")
|
||||
}
|
||||
return nil
|
||||
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"fmt"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
// NoopRunner is a deterministic no-op seriatim adapter.
|
||||
@@ -260,7 +261,7 @@ func (f *FakeRunner) Render(ctx context.Context, req RenderRequest) (RenderResul
|
||||
|
||||
func materializePlaceholders(req MergeRequest) error {
|
||||
if req.OutputMergedTranscriptPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputMergedTranscriptPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputMergedTranscriptPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write merged transcript %q: %w", req.OutputMergedTranscriptPath, err)
|
||||
}
|
||||
}
|
||||
@@ -271,22 +272,22 @@ func materializePlaceholders(req MergeRequest) error {
|
||||
"input_transcript_paths": req.InputTranscriptPaths,
|
||||
"output_path": req.OutputMergedTranscriptPath,
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.ReportPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write report %q: %w", req.ReportPath, err)
|
||||
}
|
||||
}
|
||||
@@ -295,7 +296,7 @@ func materializePlaceholders(req MergeRequest) error {
|
||||
|
||||
func materializeTrimPlaceholders(req TrimRequest) error {
|
||||
if req.OutputTrimmedPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputTrimmedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputTrimmedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write trimmed transcript %q: %w", req.OutputTrimmedPath, err)
|
||||
}
|
||||
}
|
||||
@@ -308,17 +309,17 @@ func materializeTrimPlaceholders(req TrimRequest) error {
|
||||
"output_path": req.OutputTrimmedPath,
|
||||
"keep_selector": req.KeepSelector,
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake trim stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake trim stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake trim stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake trim stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
@@ -327,7 +328,7 @@ func materializeTrimPlaceholders(req TrimRequest) error {
|
||||
|
||||
func materializeNormalizePlaceholders(req NormalizeRequest) error {
|
||||
if req.OutputNormalizedPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputNormalizedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputNormalizedPath, []byte(`{"schema":"seriatim.intermediate.v1","segments":[]}`), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write normalized transcript %q: %w", req.OutputNormalizedPath, err)
|
||||
}
|
||||
}
|
||||
@@ -343,22 +344,22 @@ func materializeNormalizePlaceholders(req NormalizeRequest) error {
|
||||
if req.ReportPath != "" {
|
||||
payload["report_path"] = req.ReportPath
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake normalize stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake normalize stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake normalize stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake normalize stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.ReportPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.ReportPath, []byte(`{"schema":"seriatim.report.v1","placeholder":true}`), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write report %q: %w", req.ReportPath, err)
|
||||
}
|
||||
}
|
||||
@@ -367,7 +368,7 @@ func materializeNormalizePlaceholders(req NormalizeRequest) error {
|
||||
|
||||
func materializeRenderPlaceholders(req RenderRequest) error {
|
||||
if req.OutputRenderedPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputRenderedPath, []byte("# Transcript\n\nRendered markdown placeholder.\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.OutputRenderedPath, []byte("# Transcript\n\nRendered markdown placeholder.\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write rendered transcript %q: %w", req.OutputRenderedPath, err)
|
||||
}
|
||||
}
|
||||
@@ -384,17 +385,17 @@ func materializeRenderPlaceholders(req RenderRequest) error {
|
||||
"include_segment_ids": req.IncludeSegmentIDs,
|
||||
"include_metadata": req.IncludeMetadata,
|
||||
}
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644); err != nil {
|
||||
if err := subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write generated config %q: %w", req.GeneratedConfigPath, err)
|
||||
}
|
||||
}
|
||||
if req.StdoutLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake render stdout placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StdoutLogPath, []byte("seriatim noop/fake render stdout placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stdout log %q: %w", req.StdoutLogPath, err)
|
||||
}
|
||||
}
|
||||
if req.StderrLogPath != "" {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake render stderr placeholder\n"), 0o644); err != nil {
|
||||
if err := subprocess.WriteFileAtomic(req.StderrLogPath, []byte("seriatim noop/fake render stderr placeholder\n"), fileops.WorkspaceFileMode); err != nil {
|
||||
return fmt.Errorf("write stderr log %q: %w", req.StderrLogPath, err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,15 +4,18 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
"unicode/utf8"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/subprocess"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
// MaxOutputFileBytes bounds each Seriatim JSON or rendered-text result.
|
||||
const MaxOutputFileBytes int64 = 64 * 1024 * 1024
|
||||
|
||||
// EnvConfig defines optional Seriatim environment tuning values.
|
||||
type EnvConfig struct {
|
||||
OverlapWordRunGap *float64
|
||||
@@ -129,12 +132,13 @@ func (r *SubprocessRunner) Run(ctx context.Context, req MergeRequest) (MergeResu
|
||||
}
|
||||
|
||||
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
||||
Executable: r.binary,
|
||||
Args: args,
|
||||
Timeout: r.timeout,
|
||||
EnvOverrides: env,
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
Executable: r.binary,
|
||||
Args: args,
|
||||
Timeout: r.timeout,
|
||||
EnvOverrides: env,
|
||||
DiagnosticOwner: "seriatim",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
if err != nil {
|
||||
return MergeResult{
|
||||
@@ -232,11 +236,12 @@ func (r *SubprocessRunner) Trim(ctx context.Context, req TrimRequest) (TrimResul
|
||||
}
|
||||
|
||||
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
||||
Executable: binary,
|
||||
Args: args,
|
||||
Timeout: timeout,
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
Executable: binary,
|
||||
Args: args,
|
||||
Timeout: timeout,
|
||||
DiagnosticOwner: "seriatim",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
if err != nil {
|
||||
return TrimResult{
|
||||
@@ -320,11 +325,12 @@ func (r *SubprocessRunner) Normalize(ctx context.Context, req NormalizeRequest)
|
||||
}
|
||||
|
||||
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
||||
Executable: binary,
|
||||
Args: args,
|
||||
Timeout: timeout,
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
Executable: binary,
|
||||
Args: args,
|
||||
Timeout: timeout,
|
||||
DiagnosticOwner: "seriatim",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
if err != nil {
|
||||
return NormalizeResult{
|
||||
@@ -425,11 +431,12 @@ func (r *SubprocessRunner) Render(ctx context.Context, req RenderRequest) (Rende
|
||||
}
|
||||
|
||||
runRes, err := subprocess.Run(ctx, subprocess.RunRequest{
|
||||
Executable: binary,
|
||||
Args: args,
|
||||
Timeout: timeout,
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
Executable: binary,
|
||||
Args: args,
|
||||
Timeout: timeout,
|
||||
DiagnosticOwner: "seriatim",
|
||||
StdoutLogPath: req.StdoutLogPath,
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
})
|
||||
if err != nil {
|
||||
return RenderResult{
|
||||
@@ -546,7 +553,7 @@ func (r *SubprocessRunner) writeMergeInvocationConfig(req MergeRequest, args []s
|
||||
payload["coalesce_gap"] = *r.coalesceGap
|
||||
}
|
||||
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
func buildTrimArgs(req TrimRequest) []string {
|
||||
@@ -598,7 +605,7 @@ func writeTrimInvocationConfig(req TrimRequest, args []string, binary string, ti
|
||||
"output_path": req.OutputTrimmedPath,
|
||||
"keep_selector": req.KeepSelector,
|
||||
}
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
func writeNormalizeInvocationConfig(req NormalizeRequest, args []string, binary string, timeout time.Duration, outputSchema string) error {
|
||||
@@ -613,7 +620,7 @@ func writeNormalizeInvocationConfig(req NormalizeRequest, args []string, binary
|
||||
"output_schema": outputSchema,
|
||||
"report_path": req.ReportPath,
|
||||
}
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
func writeRenderInvocationConfig(req RenderRequest, args []string, binary string, timeout time.Duration, format string) error {
|
||||
@@ -631,13 +638,13 @@ func writeRenderInvocationConfig(req RenderRequest, args []string, binary string
|
||||
"include_segment_ids": req.IncludeSegmentIDs,
|
||||
"include_metadata": req.IncludeMetadata,
|
||||
}
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, 0o644)
|
||||
return subprocess.WriteYAMLAtomic(req.GeneratedConfigPath, payload, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
func validateJSONFile(path string) error {
|
||||
data, err := os.ReadFile(path)
|
||||
data, err := readSeriatimResult(path, "JSON output")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read file: %w", err)
|
||||
return err
|
||||
}
|
||||
var v any
|
||||
if err := json.Unmarshal(data, &v); err != nil {
|
||||
@@ -647,9 +654,9 @@ func validateJSONFile(path string) error {
|
||||
}
|
||||
|
||||
func validateJSONFileWithSegments(path string) error {
|
||||
data, err := os.ReadFile(path)
|
||||
data, err := readSeriatimResult(path, "transcript JSON output")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read file: %w", err)
|
||||
return err
|
||||
}
|
||||
|
||||
var payload map[string]any
|
||||
@@ -668,9 +675,9 @@ func validateJSONFileWithSegments(path string) error {
|
||||
}
|
||||
|
||||
func validateNonEmptyTextFile(path string) error {
|
||||
data, err := os.ReadFile(path)
|
||||
data, err := readSeriatimResult(path, "rendered text output")
|
||||
if err != nil {
|
||||
return fmt.Errorf("read file: %w", err)
|
||||
return err
|
||||
}
|
||||
if len(data) == 0 {
|
||||
return fmt.Errorf("file is empty")
|
||||
@@ -683,3 +690,11 @@ func validateNonEmptyTextFile(path string) error {
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func readSeriatimResult(path, category string) ([]byte, error) {
|
||||
data, err := fileops.ReadRegularFile(path, MaxOutputFileBytes)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("seriatim %s exceeds or cannot be read within %d-byte limit: %w", category, MaxOutputFileBytes, err)
|
||||
}
|
||||
return data, nil
|
||||
}
|
||||
|
||||
90
internal/adapters/storage/bounded_read.go
Normal file
90
internal/adapters/storage/bounded_read.go
Normal file
@@ -0,0 +1,90 @@
|
||||
package storage
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"math"
|
||||
"strings"
|
||||
)
|
||||
|
||||
// ReadLimitError reports that a remote object exceeded its caller-owned read
|
||||
// limit. The limit is enforced against both available object metadata and the
|
||||
// bytes returned by the opened object body.
|
||||
type ReadLimitError struct {
|
||||
Key string
|
||||
Limit int64
|
||||
Observed int64
|
||||
}
|
||||
|
||||
func (e *ReadLimitError) Error() string {
|
||||
return fmt.Sprintf("object %q exceeds %d-byte read limit (observed at least %d bytes)", e.Key, e.Limit, e.Observed)
|
||||
}
|
||||
|
||||
// ReadObjectBounded opens one object version and retains at most maxBytes of
|
||||
// its content. Object metadata may reject an oversized body early, but a
|
||||
// limit-plus-one read always enforces the boundary when transfer begins.
|
||||
func ReadObjectBounded(ctx context.Context, store ObjectStore, key string, maxBytes int64) (info ObjectInfo, data []byte, err error) {
|
||||
key = strings.TrimSpace(key)
|
||||
if store == nil {
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read bounded object: store is required")
|
||||
}
|
||||
if key == "" {
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read bounded object: key is required")
|
||||
}
|
||||
if maxBytes <= 0 || maxBytes == math.MaxInt64 {
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read bounded object %q: limit must be between 1 and %d bytes", key, int64(math.MaxInt64-1))
|
||||
}
|
||||
if err := ctx.Err(); err != nil {
|
||||
return ObjectInfo{}, nil, err
|
||||
}
|
||||
|
||||
info, body, err := store.Read(ctx, key)
|
||||
if err != nil {
|
||||
return ObjectInfo{}, nil, err
|
||||
}
|
||||
if body == nil {
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read bounded object %q: store returned no body", key)
|
||||
}
|
||||
defer func() {
|
||||
if closeErr := body.Close(); closeErr != nil {
|
||||
data = nil
|
||||
err = errors.Join(err, fmt.Errorf("close object %q: %w", key, closeErr))
|
||||
}
|
||||
}()
|
||||
|
||||
if info.Size > maxBytes {
|
||||
return info, nil, &ReadLimitError{Key: key, Limit: maxBytes, Observed: info.Size}
|
||||
}
|
||||
|
||||
data, err = io.ReadAll(io.LimitReader(contextReader{ctx: ctx, reader: body}, maxBytes+1))
|
||||
if err != nil {
|
||||
return info, nil, err
|
||||
}
|
||||
if err := ctx.Err(); err != nil {
|
||||
return info, nil, err
|
||||
}
|
||||
if int64(len(data)) > maxBytes {
|
||||
return info, nil, &ReadLimitError{Key: key, Limit: maxBytes, Observed: int64(len(data))}
|
||||
}
|
||||
return info, data, nil
|
||||
}
|
||||
|
||||
type contextReader struct {
|
||||
ctx context.Context
|
||||
reader io.Reader
|
||||
}
|
||||
|
||||
func (r contextReader) Read(p []byte) (int, error) {
|
||||
if err := r.ctx.Err(); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
n, err := r.reader.Read(p)
|
||||
if err == nil {
|
||||
if contextErr := r.ctx.Err(); contextErr != nil {
|
||||
return n, contextErr
|
||||
}
|
||||
}
|
||||
return n, err
|
||||
}
|
||||
135
internal/adapters/storage/bounded_read_test.go
Normal file
135
internal/adapters/storage/bounded_read_test.go
Normal file
@@ -0,0 +1,135 @@
|
||||
package storage
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestReadObjectBoundedAcceptsExactLimitWithAbsentSizeMetadata(t *testing.T) {
|
||||
body := &trackingReadCloser{reader: bytes.NewReader([]byte("12345678")), chunkSize: 2}
|
||||
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||
return ObjectInfo{Key: "control.json", ETag: "generation"}, body, nil
|
||||
}}
|
||||
|
||||
info, data, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
|
||||
if err != nil {
|
||||
t.Fatalf("ReadObjectBounded() error = %v", err)
|
||||
}
|
||||
if string(data) != "12345678" || info.ETag != "generation" {
|
||||
t.Fatalf("ReadObjectBounded() = (%#v, %q), want opened object metadata and bytes", info, data)
|
||||
}
|
||||
if !body.closed {
|
||||
t.Fatal("object body was not closed")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadObjectBoundedRejectsLimitPlusOneDespiteMissingOrInaccurateMetadata(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
metadataSize int64
|
||||
}{
|
||||
{name: "missing", metadataSize: 0},
|
||||
{name: "inaccurate", metadataSize: 2},
|
||||
}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
body := &trackingReadCloser{reader: bytes.NewReader([]byte("123456789")), chunkSize: 1}
|
||||
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||
return ObjectInfo{Key: "control.json", Size: test.metadataSize}, body, nil
|
||||
}}
|
||||
|
||||
_, data, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
|
||||
var limitErr *ReadLimitError
|
||||
if !errors.As(err, &limitErr) {
|
||||
t.Fatalf("ReadObjectBounded() error = %v, want ReadLimitError", err)
|
||||
}
|
||||
if data != nil || body.bytesRead != 9 || !body.closed {
|
||||
t.Fatalf("data=%q bytes read=%d closed=%t, want nil, 9, true", data, body.bytesRead, body.closed)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadObjectBoundedRejectsOversizedMetadataBeforeTransfer(t *testing.T) {
|
||||
body := &trackingReadCloser{reader: bytes.NewReader([]byte("small"))}
|
||||
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||
return ObjectInfo{Key: "control.json", Size: 9}, body, nil
|
||||
}}
|
||||
|
||||
_, _, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
|
||||
var limitErr *ReadLimitError
|
||||
if !errors.As(err, &limitErr) {
|
||||
t.Fatalf("ReadObjectBounded() error = %v, want ReadLimitError", err)
|
||||
}
|
||||
if body.bytesRead != 0 || !body.closed {
|
||||
t.Fatalf("bytes read=%d closed=%t, want zero-byte transfer and closed body", body.bytesRead, body.closed)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadObjectBoundedPropagatesCancellationAndClosesBody(t *testing.T) {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
body := &trackingReadCloser{reader: bytes.NewReader([]byte("12345678")), chunkSize: 1, afterRead: cancel}
|
||||
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||
return ObjectInfo{Key: "control.json"}, body, nil
|
||||
}}
|
||||
|
||||
_, data, err := ReadObjectBounded(ctx, store, "control.json", 8)
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("ReadObjectBounded() error = %v, want context cancellation", err)
|
||||
}
|
||||
if data != nil || body.bytesRead != 1 || !body.closed {
|
||||
t.Fatalf("data=%q bytes read=%d closed=%t, want nil, 1, true", data, body.bytesRead, body.closed)
|
||||
}
|
||||
}
|
||||
|
||||
func TestReadObjectBoundedReturnsCloseFailure(t *testing.T) {
|
||||
closeErr := errors.New("close failed")
|
||||
body := &trackingReadCloser{reader: bytes.NewReader([]byte("ok")), closeErr: closeErr}
|
||||
store := &boundedReadStore{read: func(context.Context, string) (ObjectInfo, io.ReadCloser, error) {
|
||||
return ObjectInfo{Key: "control.json", Size: 2}, body, nil
|
||||
}}
|
||||
|
||||
_, data, err := ReadObjectBounded(context.Background(), store, "control.json", 8)
|
||||
if !errors.Is(err, closeErr) || data != nil || !body.closed {
|
||||
t.Fatalf("data=%q error=%v closed=%t, want close failure and no retained data", data, err, body.closed)
|
||||
}
|
||||
}
|
||||
|
||||
type boundedReadStore struct {
|
||||
ObjectStore
|
||||
read func(context.Context, string) (ObjectInfo, io.ReadCloser, error)
|
||||
}
|
||||
|
||||
func (s *boundedReadStore) Read(ctx context.Context, key string) (ObjectInfo, io.ReadCloser, error) {
|
||||
return s.read(ctx, key)
|
||||
}
|
||||
|
||||
type trackingReadCloser struct {
|
||||
reader io.Reader
|
||||
chunkSize int
|
||||
afterRead func()
|
||||
closeErr error
|
||||
bytesRead int
|
||||
closed bool
|
||||
}
|
||||
|
||||
func (r *trackingReadCloser) Read(p []byte) (int, error) {
|
||||
if r.chunkSize > 0 && len(p) > r.chunkSize {
|
||||
p = p[:r.chunkSize]
|
||||
}
|
||||
n, err := r.reader.Read(p)
|
||||
r.bytesRead += n
|
||||
if n > 0 && r.afterRead != nil {
|
||||
r.afterRead()
|
||||
r.afterRead = nil
|
||||
}
|
||||
return n, err
|
||||
}
|
||||
|
||||
func (r *trackingReadCloser) Close() error {
|
||||
r.closed = true
|
||||
return r.closeErr
|
||||
}
|
||||
23
internal/adapters/storage/download_writer.go
Normal file
23
internal/adapters/storage/download_writer.go
Normal file
@@ -0,0 +1,23 @@
|
||||
package storage
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
)
|
||||
|
||||
// WriterDownloader is implemented by storage backends that stream an object
|
||||
// into a caller-owned file handle.
|
||||
type WriterDownloader interface {
|
||||
DownloadTo(ctx context.Context, key string, destination io.Writer) error
|
||||
}
|
||||
|
||||
// DownloadTo streams one object into destination. Destination-confined callers
|
||||
// require this capability rather than granting a backend a mutable pathname.
|
||||
func DownloadTo(ctx context.Context, store ObjectStore, key string, destination io.Writer) error {
|
||||
writer, ok := store.(WriterDownloader)
|
||||
if !ok {
|
||||
return fmt.Errorf("object store does not support handle-confined downloads")
|
||||
}
|
||||
return writer.DownloadTo(ctx, key, destination)
|
||||
}
|
||||
@@ -14,16 +14,15 @@ func NewObjectStoreFromConfig(ctx context.Context, cfg *config.Config) (ObjectSt
|
||||
return nil, fmt.Errorf("pipeline config is required")
|
||||
}
|
||||
|
||||
if strings.EqualFold(strings.TrimSpace(cfg.Pipeline.Storage.Backend), "s3") {
|
||||
switch strings.ToLower(strings.TrimSpace(cfg.Pipeline.Storage.Backend)) {
|
||||
case config.StorageBackendS3:
|
||||
if cfg.Pipeline.Storage.S3 == nil {
|
||||
return nil, fmt.Errorf("pipeline.storage.s3 is required when pipeline.storage.backend is s3")
|
||||
}
|
||||
return NewS3BackendFromConfig(ctx, *cfg.Pipeline.Storage.S3)
|
||||
case "", config.StorageBackendLocal:
|
||||
return nil, fmt.Errorf("no remote object store backend is configured")
|
||||
default:
|
||||
return nil, fmt.Errorf("unsupported pipeline.storage.backend %q", cfg.Pipeline.Storage.Backend)
|
||||
}
|
||||
|
||||
if cfg.Pipeline.Storage.S3 != nil && strings.TrimSpace(cfg.Pipeline.Storage.S3.Bucket) != "" {
|
||||
return NewS3BackendFromConfig(ctx, *cfg.Pipeline.Storage.S3)
|
||||
}
|
||||
|
||||
return nil, fmt.Errorf("no remote object store backend is configured")
|
||||
}
|
||||
|
||||
@@ -54,3 +54,35 @@ func TestNewObjectStoreFromConfigNoRemoteBackendConfigured(t *testing.T) {
|
||||
t.Fatalf("NewObjectStoreFromConfig() error = %v, want no-backend error", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewObjectStoreFromConfigDoesNotInferS3FromProviderFields(t *testing.T) {
|
||||
called := false
|
||||
original := newS3Client
|
||||
t.Cleanup(func() { newS3Client = original })
|
||||
newS3Client = func(_ context.Context, _ s3ClientOptions) (s3API, error) {
|
||||
called = true
|
||||
return &fakeS3API{}, nil
|
||||
}
|
||||
|
||||
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
|
||||
Pipeline: &config.PipelineConfig{Storage: config.StorageConfig{
|
||||
Backend: config.StorageBackendLocal,
|
||||
S3: &config.StorageS3Config{Bucket: "my-archive"},
|
||||
}},
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), "no remote object store backend is configured") {
|
||||
t.Fatalf("NewObjectStoreFromConfig() error = %v, want no-backend error", err)
|
||||
}
|
||||
if called {
|
||||
t.Fatal("NewObjectStoreFromConfig() constructed S3 from incidental provider fields")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNewObjectStoreFromConfigRejectsUnknownBackend(t *testing.T) {
|
||||
_, err := NewObjectStoreFromConfig(context.Background(), &config.Config{
|
||||
Pipeline: &config.PipelineConfig{Storage: config.StorageConfig{Backend: "s33"}},
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), "unsupported pipeline.storage.backend") {
|
||||
t.Fatalf("NewObjectStoreFromConfig() error = %v, want unsupported-backend error", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,25 +1,35 @@
|
||||
package storage
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// FakeBackend provides a deterministic in-memory object store for tests.
|
||||
type FakeBackend struct {
|
||||
mu sync.RWMutex
|
||||
|
||||
Objects map[string]FakeObject
|
||||
Uploads []FakeUploadCall
|
||||
Downloads []FakeDownloadCall
|
||||
Reads []FakeReadCall
|
||||
|
||||
ListErr error
|
||||
DownloadErr error
|
||||
UploadErr error
|
||||
ExistsErr error
|
||||
ListErr error
|
||||
DownloadErr error
|
||||
UploadErr error
|
||||
ExistsErr error
|
||||
UploadHook func(FakeUploadCall) error
|
||||
DownloadHook func(FakeDownloadCall) error
|
||||
}
|
||||
|
||||
// FakeUploadCall captures one upload invocation in call order.
|
||||
@@ -33,6 +43,12 @@ type FakeUploadCall struct {
|
||||
type FakeDownloadCall struct {
|
||||
Key string
|
||||
LocalPath string
|
||||
Bytes int64
|
||||
}
|
||||
|
||||
// FakeReadCall captures one opened object in call order.
|
||||
type FakeReadCall struct {
|
||||
Key string
|
||||
}
|
||||
|
||||
// FakeObject is a deterministic fake object-store record.
|
||||
@@ -46,6 +62,12 @@ type FakeObject struct {
|
||||
|
||||
// SeedObject inserts or replaces an object in the fake object store.
|
||||
func (f *FakeBackend) SeedObject(obj FakeObject) {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
f.seedObject(obj)
|
||||
}
|
||||
|
||||
func (f *FakeBackend) seedObject(obj FakeObject) {
|
||||
if f.Objects == nil {
|
||||
f.Objects = map[string]FakeObject{}
|
||||
}
|
||||
@@ -53,6 +75,9 @@ func (f *FakeBackend) SeedObject(obj FakeObject) {
|
||||
obj.Key = key
|
||||
obj.Data = append([]byte(nil), obj.Data...)
|
||||
obj.Metadata = copyMetadata(obj.Metadata)
|
||||
if obj.ETag == "" {
|
||||
obj.ETag = fakeObjectETag(obj.Data)
|
||||
}
|
||||
f.Objects[key] = obj
|
||||
}
|
||||
|
||||
@@ -65,6 +90,8 @@ func (f *FakeBackend) List(ctx context.Context, prefix string) ([]ObjectInfo, er
|
||||
return nil, f.ListErr
|
||||
}
|
||||
|
||||
f.mu.RLock()
|
||||
defer f.mu.RUnlock()
|
||||
normalizedPrefix := normalizeObjectKey(prefix)
|
||||
keys := make([]string, 0, len(f.Objects))
|
||||
for key := range f.Objects {
|
||||
@@ -87,34 +114,77 @@ func (f *FakeBackend) List(ctx context.Context, prefix string) ([]ObjectInfo, er
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Download writes one object to a local path.
|
||||
func (f *FakeBackend) Download(ctx context.Context, key, localPath string) error {
|
||||
// Read returns a stable object body and the generation observed with it.
|
||||
func (f *FakeBackend) Read(ctx context.Context, key string) (ObjectInfo, io.ReadCloser, error) {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return ObjectInfo{}, nil, err
|
||||
}
|
||||
if f.DownloadErr != nil {
|
||||
return ObjectInfo{}, nil, f.DownloadErr
|
||||
}
|
||||
normalizedKey := normalizeObjectKey(key)
|
||||
f.mu.RLock()
|
||||
obj, ok := f.Objects[normalizedKey]
|
||||
if ok {
|
||||
obj.Data = append([]byte(nil), obj.Data...)
|
||||
obj.Metadata = copyMetadata(obj.Metadata)
|
||||
}
|
||||
f.mu.RUnlock()
|
||||
if !ok {
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read object %q: %w", normalizedKey, os.ErrNotExist)
|
||||
}
|
||||
f.mu.Lock()
|
||||
f.Reads = append(f.Reads, FakeReadCall{Key: normalizedKey})
|
||||
f.mu.Unlock()
|
||||
return ObjectInfo{Key: obj.Key, Size: int64(len(obj.Data)), ETag: obj.ETag, LastModified: obj.LastModified}, io.NopCloser(bytes.NewReader(obj.Data)), nil
|
||||
}
|
||||
|
||||
// DownloadTo writes one object to a caller-owned destination writer.
|
||||
func (f *FakeBackend) DownloadTo(ctx context.Context, key string, destination io.Writer) error {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return err
|
||||
}
|
||||
if f.DownloadErr != nil {
|
||||
return f.DownloadErr
|
||||
}
|
||||
if destination == nil {
|
||||
return fmt.Errorf("download object: destination writer is required")
|
||||
}
|
||||
if f.DownloadHook != nil {
|
||||
if err := f.DownloadHook(FakeDownloadCall{Key: normalizeObjectKey(key)}); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
_, source, err := f.Read(ctx, key)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer source.Close()
|
||||
count, err := io.Copy(destination, source)
|
||||
if err != nil {
|
||||
return fmt.Errorf("download object %q: write destination: %w", key, err)
|
||||
}
|
||||
f.mu.Lock()
|
||||
f.Downloads = append(f.Downloads, FakeDownloadCall{Key: normalizeObjectKey(key), Bytes: count})
|
||||
f.mu.Unlock()
|
||||
return nil
|
||||
}
|
||||
|
||||
// Download writes one object to a local path.
|
||||
func (f *FakeBackend) Download(ctx context.Context, key, localPath string) error {
|
||||
if strings.TrimSpace(localPath) == "" {
|
||||
return fmt.Errorf("download object: local path is required")
|
||||
}
|
||||
|
||||
obj, ok := f.Objects[normalizeObjectKey(key)]
|
||||
if !ok {
|
||||
return fmt.Errorf("download object %q: %w", key, os.ErrNotExist)
|
||||
}
|
||||
f.Downloads = append(f.Downloads, FakeDownloadCall{
|
||||
Key: normalizeObjectKey(key),
|
||||
LocalPath: localPath,
|
||||
})
|
||||
|
||||
if err := os.MkdirAll(filepath.Dir(localPath), 0o755); err != nil {
|
||||
return fmt.Errorf("download object %q: create parent directory: %w", key, err)
|
||||
}
|
||||
if err := os.WriteFile(localPath, obj.Data, 0o644); err != nil {
|
||||
return fmt.Errorf("download object %q: write local file: %w", key, err)
|
||||
destination, err := os.Create(localPath)
|
||||
if err != nil {
|
||||
return fmt.Errorf("download object %q: create local file: %w", key, err)
|
||||
}
|
||||
return nil
|
||||
defer destination.Close()
|
||||
return f.DownloadTo(ctx, key, destination)
|
||||
}
|
||||
|
||||
// Upload reads a local file and stores it under key.
|
||||
@@ -132,33 +202,117 @@ func (f *FakeBackend) Upload(ctx context.Context, localPath, key string, opts Up
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
|
||||
}
|
||||
|
||||
data, err := os.ReadFile(localPath)
|
||||
file, err := os.Open(localPath)
|
||||
if err != nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", key, localPath, err)
|
||||
}
|
||||
defer file.Close()
|
||||
return f.uploadReader(ctx, file, key, opts, localPath)
|
||||
}
|
||||
|
||||
// UploadReader stores content provided by a caller-owned reader.
|
||||
func (f *FakeBackend) UploadReader(ctx context.Context, source io.Reader, key string, opts UploadOptions) (ObjectInfo, error) {
|
||||
return f.uploadReader(ctx, source, key, opts, "reader")
|
||||
}
|
||||
|
||||
func (f *FakeBackend) uploadReader(ctx context.Context, source io.Reader, key string, opts UploadOptions, localPath string) (ObjectInfo, error) {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return ObjectInfo{}, err
|
||||
}
|
||||
if f.UploadErr != nil {
|
||||
return ObjectInfo{}, f.UploadErr
|
||||
}
|
||||
if source == nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: source is required")
|
||||
}
|
||||
if strings.TrimSpace(key) == "" {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
|
||||
}
|
||||
|
||||
data, err := io.ReadAll(source)
|
||||
if err != nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", key, localPath, err)
|
||||
}
|
||||
|
||||
normalizedKey := normalizeObjectKey(key)
|
||||
f.Uploads = append(f.Uploads, FakeUploadCall{
|
||||
call := FakeUploadCall{
|
||||
LocalPath: localPath,
|
||||
Key: normalizedKey,
|
||||
Options: UploadOptions{
|
||||
Metadata: copyMetadata(opts.Metadata),
|
||||
ContentType: opts.ContentType,
|
||||
},
|
||||
})
|
||||
now := time.Now().UTC()
|
||||
obj := FakeObject{
|
||||
Key: normalizedKey,
|
||||
Data: data,
|
||||
Metadata: copyMetadata(opts.Metadata),
|
||||
LastModified: &now,
|
||||
}
|
||||
f.SeedObject(obj)
|
||||
return ObjectInfo{
|
||||
Key: normalizedKey,
|
||||
Size: int64(len(data)),
|
||||
LastModified: &now,
|
||||
}, nil
|
||||
f.mu.Lock()
|
||||
f.Uploads = append(f.Uploads, call)
|
||||
f.mu.Unlock()
|
||||
if f.UploadHook != nil {
|
||||
if err := f.UploadHook(call); err != nil {
|
||||
return ObjectInfo{}, err
|
||||
}
|
||||
}
|
||||
return f.storeUploadedObject(normalizedKey, data, opts), nil
|
||||
}
|
||||
|
||||
// UploadConditional atomically checks and replaces one mutable object.
|
||||
func (f *FakeBackend) UploadConditional(ctx context.Context, source io.Reader, key string, opts UploadOptions, condition WriteCondition) (ObjectInfo, error) {
|
||||
if err := ctx.Err(); err != nil {
|
||||
return ObjectInfo{}, err
|
||||
}
|
||||
if err := validateWriteCondition(condition); err != nil {
|
||||
return ObjectInfo{}, err
|
||||
}
|
||||
if f.UploadErr != nil {
|
||||
return ObjectInfo{}, f.UploadErr
|
||||
}
|
||||
if source == nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: source is required")
|
||||
}
|
||||
normalizedKey := normalizeObjectKey(key)
|
||||
if normalizedKey == "" {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
|
||||
}
|
||||
data, err := io.ReadAll(source)
|
||||
if err != nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q: %w", normalizedKey, err)
|
||||
}
|
||||
call := FakeUploadCall{Key: normalizedKey, Options: UploadOptions{Metadata: copyMetadata(opts.Metadata), ContentType: opts.ContentType}}
|
||||
f.mu.Lock()
|
||||
f.Uploads = append(f.Uploads, call)
|
||||
f.mu.Unlock()
|
||||
if f.UploadHook != nil {
|
||||
if err := f.UploadHook(call); err != nil {
|
||||
return ObjectInfo{}, err
|
||||
}
|
||||
}
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
existing, found := f.Objects[normalizedKey]
|
||||
if condition.RequireAbsent && found {
|
||||
return ObjectInfo{}, ErrConditionNotMet
|
||||
}
|
||||
if expected := strings.TrimSpace(condition.MatchETag); expected != "" && (!found || existing.ETag != expected) {
|
||||
return ObjectInfo{}, ErrConditionNotMet
|
||||
}
|
||||
return f.storeUploadedObjectLocked(normalizedKey, data, opts), nil
|
||||
}
|
||||
|
||||
func (f *FakeBackend) storeUploadedObject(key string, data []byte, opts UploadOptions) ObjectInfo {
|
||||
f.mu.Lock()
|
||||
defer f.mu.Unlock()
|
||||
return f.storeUploadedObjectLocked(key, data, opts)
|
||||
}
|
||||
|
||||
func (f *FakeBackend) storeUploadedObjectLocked(key string, data []byte, opts UploadOptions) ObjectInfo {
|
||||
now := time.Now().UTC()
|
||||
obj := FakeObject{Key: key, Data: append([]byte(nil), data...), Metadata: copyMetadata(opts.Metadata), LastModified: &now}
|
||||
f.seedObject(obj)
|
||||
return ObjectInfo{Key: key, Size: int64(len(data)), ETag: fakeObjectETag(data), LastModified: &now}
|
||||
}
|
||||
|
||||
func fakeObjectETag(data []byte) string {
|
||||
sum := sha256.Sum256(data)
|
||||
return hex.EncodeToString(sum[:])
|
||||
}
|
||||
|
||||
// Exists checks object presence.
|
||||
@@ -169,7 +323,9 @@ func (f *FakeBackend) Exists(ctx context.Context, key string) (bool, error) {
|
||||
if f.ExistsErr != nil {
|
||||
return false, f.ExistsErr
|
||||
}
|
||||
f.mu.RLock()
|
||||
_, ok := f.Objects[normalizeObjectKey(key)]
|
||||
f.mu.RUnlock()
|
||||
return ok, nil
|
||||
}
|
||||
|
||||
|
||||
@@ -71,6 +71,21 @@ func TestFakeBackendUploadAndExists(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestFakeBackendConditionalUploadRejectsStaleGeneration(t *testing.T) {
|
||||
fake := &FakeBackend{}
|
||||
fake.SeedObject(FakeObject{Key: "locks.yml", Data: []byte("old")})
|
||||
old := fake.Objects["locks.yml"].ETag
|
||||
if _, err := fake.UploadConditional(context.Background(), strings.NewReader("new"), "locks.yml", UploadOptions{}, WriteCondition{MatchETag: old}); err != nil {
|
||||
t.Fatalf("UploadConditional() error = %v", err)
|
||||
}
|
||||
if _, err := fake.UploadConditional(context.Background(), strings.NewReader("lost"), "locks.yml", UploadOptions{}, WriteCondition{MatchETag: old}); !errors.Is(err, ErrConditionNotMet) {
|
||||
t.Fatalf("UploadConditional() error = %v, want ErrConditionNotMet", err)
|
||||
}
|
||||
if got := string(fake.Objects["locks.yml"].Data); got != "new" {
|
||||
t.Fatalf("locks object = %q, want successful replacement preserved", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFakeBackendObjectErrors(t *testing.T) {
|
||||
fake := &FakeBackend{DownloadErr: errors.New("download fail"), UploadErr: errors.New("upload fail"), ListErr: errors.New("list fail"), ExistsErr: errors.New("exists fail")}
|
||||
|
||||
|
||||
@@ -2,9 +2,21 @@ package storage
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"time"
|
||||
)
|
||||
|
||||
// ErrConditionNotMet reports that an object changed or already existed before a
|
||||
// conditional write could be committed.
|
||||
var ErrConditionNotMet = errors.New("object write condition not met")
|
||||
|
||||
// ReaderUploader streams caller-owned, already-opened content to object storage.
|
||||
// Callers retain source-selection and filesystem-confinement policy.
|
||||
type ReaderUploader interface {
|
||||
UploadReader(ctx context.Context, source io.Reader, key string, opts UploadOptions) (ObjectInfo, error)
|
||||
}
|
||||
|
||||
// ObjectStore is a remote object storage boundary used by prepare, restore, and publish work.
|
||||
//
|
||||
// Key invariant:
|
||||
@@ -12,8 +24,10 @@ import (
|
||||
// infer Narratio session semantics and do not prepend root prefixes.
|
||||
type ObjectStore interface {
|
||||
List(ctx context.Context, prefix string) ([]ObjectInfo, error)
|
||||
Read(ctx context.Context, key string) (ObjectInfo, io.ReadCloser, error)
|
||||
Download(ctx context.Context, key, localPath string) error
|
||||
Upload(ctx context.Context, localPath, key string, opts UploadOptions) (ObjectInfo, error)
|
||||
UploadConditional(ctx context.Context, source io.Reader, key string, opts UploadOptions, condition WriteCondition) (ObjectInfo, error)
|
||||
Exists(ctx context.Context, key string) (bool, error)
|
||||
}
|
||||
|
||||
@@ -30,3 +44,10 @@ type UploadOptions struct {
|
||||
Metadata map[string]string
|
||||
ContentType string
|
||||
}
|
||||
|
||||
// WriteCondition protects a mutable object update against a stale snapshot.
|
||||
// Exactly one condition is required by UploadConditional.
|
||||
type WriteCondition struct {
|
||||
MatchETag string
|
||||
RequireAbsent bool
|
||||
}
|
||||
|
||||
@@ -117,6 +117,7 @@ func (b *S3Backend) List(ctx context.Context, prefix string) ([]ObjectInfo, erro
|
||||
normalizedPrefix := normalizeObjectKey(prefix)
|
||||
out := make([]ObjectInfo, 0)
|
||||
var token *string
|
||||
seenTokens := map[string]struct{}{}
|
||||
|
||||
for {
|
||||
resp, err := b.client.ListObjectsV2(ctx, &s3.ListObjectsV2Input{
|
||||
@@ -142,44 +143,78 @@ func (b *S3Backend) List(ctx context.Context, prefix string) ([]ObjectInfo, erro
|
||||
})
|
||||
}
|
||||
|
||||
if !valueOrFalseBool(resp.IsTruncated) || resp.NextContinuationToken == nil {
|
||||
if !valueOrFalseBool(resp.IsTruncated) {
|
||||
break
|
||||
}
|
||||
token = resp.NextContinuationToken
|
||||
next := strings.TrimSpace(valueOrEmpty(resp.NextContinuationToken))
|
||||
if next == "" {
|
||||
return nil, fmt.Errorf("s3 list objects bucket %q prefix %q: truncated response has an empty continuation token", b.bucket, normalizedPrefix)
|
||||
}
|
||||
if _, repeated := seenTokens[next]; repeated {
|
||||
return nil, fmt.Errorf("s3 list objects bucket %q prefix %q: truncated response repeated continuation token", b.bucket, normalizedPrefix)
|
||||
}
|
||||
seenTokens[next] = struct{}{}
|
||||
token = &next
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// Read retrieves an object together with the generation observed for its body.
|
||||
func (b *S3Backend) Read(ctx context.Context, key string) (ObjectInfo, io.ReadCloser, error) {
|
||||
normalizedKey := normalizeObjectKey(key)
|
||||
resp, err := b.client.GetObject(ctx, &s3.GetObjectInput{Bucket: &b.bucket, Key: &normalizedKey})
|
||||
if err != nil {
|
||||
if isS3NotFound(err) {
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read object %q: %w", normalizedKey, os.ErrNotExist)
|
||||
}
|
||||
return ObjectInfo{}, nil, fmt.Errorf("read object %q: %w", normalizedKey, err)
|
||||
}
|
||||
var lastModified *time.Time
|
||||
if resp.LastModified != nil {
|
||||
t := *resp.LastModified
|
||||
lastModified = &t
|
||||
}
|
||||
return ObjectInfo{
|
||||
Key: normalizedKey, Size: valueOrZeroInt64(resp.ContentLength),
|
||||
ETag: strings.Trim(valueOrEmpty(resp.ETag), "\""), LastModified: lastModified,
|
||||
}, resp.Body, nil
|
||||
}
|
||||
|
||||
// DownloadTo retrieves one object into the caller-owned destination writer.
|
||||
func (b *S3Backend) DownloadTo(ctx context.Context, key string, destination io.Writer) error {
|
||||
if destination == nil {
|
||||
return fmt.Errorf("download object: destination writer is required")
|
||||
}
|
||||
_, body, err := b.Read(ctx, key)
|
||||
if err != nil {
|
||||
return fmt.Errorf("download object %q: %w", normalizeObjectKey(key), err)
|
||||
}
|
||||
defer body.Close()
|
||||
|
||||
if _, err := io.Copy(destination, body); err != nil {
|
||||
return fmt.Errorf("download object %q: copy body: %w", normalizeObjectKey(key), err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Download retrieves one object to localPath, creating parent directories as needed.
|
||||
func (b *S3Backend) Download(ctx context.Context, key, localPath string) error {
|
||||
normalizedKey := normalizeObjectKey(key)
|
||||
if strings.TrimSpace(localPath) == "" {
|
||||
return fmt.Errorf("download object: local path is required")
|
||||
}
|
||||
|
||||
resp, err := b.client.GetObject(ctx, &s3.GetObjectInput{
|
||||
Bucket: &b.bucket,
|
||||
Key: &normalizedKey,
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("download object %q: %w", normalizedKey, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if err := os.MkdirAll(filepath.Dir(localPath), 0o755); err != nil {
|
||||
return fmt.Errorf("download object %q: create parent directory: %w", normalizedKey, err)
|
||||
return fmt.Errorf("download object %q: create parent directory: %w", key, err)
|
||||
}
|
||||
dst, err := os.Create(localPath)
|
||||
if err != nil {
|
||||
return fmt.Errorf("download object %q: create local file: %w", normalizedKey, err)
|
||||
return fmt.Errorf("download object %q: create local file: %w", key, err)
|
||||
}
|
||||
defer dst.Close()
|
||||
|
||||
if _, err := io.Copy(dst, resp.Body); err != nil {
|
||||
return fmt.Errorf("download object %q: copy body: %w", normalizedKey, err)
|
||||
if err := b.DownloadTo(ctx, key, dst); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := dst.Sync(); err != nil {
|
||||
return fmt.Errorf("download object %q: sync local file: %w", normalizedKey, err)
|
||||
return fmt.Errorf("download object %q: sync local file: %w", key, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -204,28 +239,65 @@ func (b *S3Backend) Upload(ctx context.Context, localPath, key string, opts Uplo
|
||||
if err != nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: stat local file: %w", normalizedKey, localPath, err)
|
||||
}
|
||||
return b.uploadReader(ctx, file, key, opts, stat.Size(), WriteCondition{})
|
||||
}
|
||||
|
||||
// UploadReader sends caller-owned content to key.
|
||||
func (b *S3Backend) UploadReader(ctx context.Context, source io.Reader, key string, opts UploadOptions) (ObjectInfo, error) {
|
||||
return b.uploadReader(ctx, source, key, opts, 0, WriteCondition{})
|
||||
}
|
||||
|
||||
// UploadConditional uploads a mutable object only when its observed generation
|
||||
// still matches, or when no object exists yet.
|
||||
func (b *S3Backend) UploadConditional(ctx context.Context, source io.Reader, key string, opts UploadOptions, condition WriteCondition) (ObjectInfo, error) {
|
||||
if err := validateWriteCondition(condition); err != nil {
|
||||
return ObjectInfo{}, err
|
||||
}
|
||||
return b.uploadReader(ctx, source, key, opts, 0, condition)
|
||||
}
|
||||
|
||||
func (b *S3Backend) uploadReader(ctx context.Context, source io.Reader, key string, opts UploadOptions, size int64, condition WriteCondition) (ObjectInfo, error) {
|
||||
normalizedKey := normalizeObjectKey(key)
|
||||
if source == nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: source is required")
|
||||
}
|
||||
if normalizedKey == "" {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object: key is required")
|
||||
}
|
||||
|
||||
input := &s3.PutObjectInput{
|
||||
Bucket: &b.bucket,
|
||||
Key: &normalizedKey,
|
||||
Body: file,
|
||||
Body: source,
|
||||
Metadata: copyMetadata(opts.Metadata),
|
||||
}
|
||||
if strings.TrimSpace(opts.ContentType) != "" {
|
||||
ct := strings.TrimSpace(opts.ContentType)
|
||||
input.ContentType = &ct
|
||||
}
|
||||
if condition.RequireAbsent {
|
||||
wildcard := "*"
|
||||
input.IfNoneMatch = &wildcard
|
||||
} else if expected := strings.TrimSpace(condition.MatchETag); expected != "" {
|
||||
input.IfMatch = &expected
|
||||
}
|
||||
|
||||
resp, err := b.client.PutObject(ctx, input)
|
||||
if err != nil {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q from %q: %w", normalizedKey, localPath, err)
|
||||
if isS3ConditionalConflict(err) {
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q: %w", normalizedKey, ErrConditionNotMet)
|
||||
}
|
||||
return ObjectInfo{}, fmt.Errorf("upload object %q: %w", normalizedKey, err)
|
||||
}
|
||||
|
||||
return ObjectInfo{
|
||||
info := ObjectInfo{
|
||||
Key: normalizedKey,
|
||||
Size: stat.Size(),
|
||||
ETag: strings.Trim(valueOrEmpty(resp.ETag), "\""),
|
||||
}, nil
|
||||
}
|
||||
if size > 0 {
|
||||
info.Size = size
|
||||
}
|
||||
return info, nil
|
||||
}
|
||||
|
||||
// Exists checks whether one object key exists.
|
||||
@@ -239,18 +311,43 @@ func (b *S3Backend) Exists(ctx context.Context, key string) (bool, error) {
|
||||
return true, nil
|
||||
}
|
||||
|
||||
if isS3NotFound(err) {
|
||||
return false, nil
|
||||
}
|
||||
return false, fmt.Errorf("head object %q: %w", normalizedKey, err)
|
||||
}
|
||||
|
||||
func validateWriteCondition(condition WriteCondition) error {
|
||||
if condition.RequireAbsent == (strings.TrimSpace(condition.MatchETag) != "") {
|
||||
return fmt.Errorf("conditional upload requires exactly one of MatchETag or RequireAbsent")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func isS3NotFound(err error) bool {
|
||||
var notFound *types.NotFound
|
||||
if errors.As(err, ¬Found) {
|
||||
return false, nil
|
||||
return true
|
||||
}
|
||||
var apiErr smithy.APIError
|
||||
if errors.As(err, &apiErr) {
|
||||
switch apiErr.ErrorCode() {
|
||||
case "NotFound", "NoSuchKey", "404":
|
||||
return false, nil
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false, fmt.Errorf("head object %q: %w", normalizedKey, err)
|
||||
return false
|
||||
}
|
||||
|
||||
func isS3ConditionalConflict(err error) bool {
|
||||
var apiErr smithy.APIError
|
||||
if errors.As(err, &apiErr) {
|
||||
switch apiErr.ErrorCode() {
|
||||
case "PreconditionFailed", "ConditionalRequestConflict", "412", "409":
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func valueOrEmpty(v *string) string {
|
||||
|
||||
@@ -2,6 +2,7 @@ package storage
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
@@ -17,34 +18,81 @@ import (
|
||||
)
|
||||
|
||||
type fakeS3API struct {
|
||||
listOut *s3.ListObjectsV2Output
|
||||
listErr error
|
||||
listOut *s3.ListObjectsV2Output
|
||||
listOutputs []*s3.ListObjectsV2Output
|
||||
listErr error
|
||||
listCalls int
|
||||
|
||||
getBody io.ReadCloser
|
||||
getErr error
|
||||
getBody io.ReadCloser
|
||||
getErr error
|
||||
getSize *int64
|
||||
getETag *string
|
||||
getLastModified *time.Time
|
||||
|
||||
putOut *s3.PutObjectOutput
|
||||
putErr error
|
||||
|
||||
headErr error
|
||||
|
||||
lastList *s3.ListObjectsV2Input
|
||||
lastGet *s3.GetObjectInput
|
||||
lastPut *s3.PutObjectInput
|
||||
lastHead *s3.HeadObjectInput
|
||||
lastList *s3.ListObjectsV2Input
|
||||
lastLists []*s3.ListObjectsV2Input
|
||||
lastGet *s3.GetObjectInput
|
||||
lastPut *s3.PutObjectInput
|
||||
lastHead *s3.HeadObjectInput
|
||||
}
|
||||
|
||||
func (f *fakeS3API) ListObjectsV2(_ context.Context, params *s3.ListObjectsV2Input, _ ...func(*s3.Options)) (*s3.ListObjectsV2Output, error) {
|
||||
f.lastList = params
|
||||
f.lastLists = append(f.lastLists, params)
|
||||
if f.listErr != nil {
|
||||
return nil, f.listErr
|
||||
}
|
||||
if f.listCalls < len(f.listOutputs) {
|
||||
out := f.listOutputs[f.listCalls]
|
||||
f.listCalls++
|
||||
return out, nil
|
||||
}
|
||||
if f.listOut == nil {
|
||||
return &s3.ListObjectsV2Output{}, nil
|
||||
}
|
||||
return f.listOut, nil
|
||||
}
|
||||
|
||||
func TestS3BackendListPaginatesAndRejectsNonProgressingTokens(t *testing.T) {
|
||||
t.Run("multiple pages", func(t *testing.T) {
|
||||
client := &fakeS3API{listOutputs: []*s3.ListObjectsV2Output{
|
||||
{Contents: []types.Object{{Key: strPtr("prefix/a"), Size: int64Ptr(1)}}, IsTruncated: boolPtr(true), NextContinuationToken: strPtr("next")},
|
||||
{Contents: []types.Object{{Key: strPtr("prefix/b"), Size: int64Ptr(2)}}, IsTruncated: boolPtr(false)},
|
||||
}}
|
||||
items, err := (&S3Backend{bucket: "bucket-1", client: client}).List(context.Background(), "prefix/")
|
||||
if err != nil {
|
||||
t.Fatalf("List() error = %v", err)
|
||||
}
|
||||
if len(items) != 2 || items[0].Key != "prefix/a" || items[1].Key != "prefix/b" {
|
||||
t.Fatalf("List() items = %#v", items)
|
||||
}
|
||||
if len(client.lastLists) != 2 || client.lastLists[1].ContinuationToken == nil || *client.lastLists[1].ContinuationToken != "next" {
|
||||
t.Fatalf("continuation calls = %#v", client.lastLists)
|
||||
}
|
||||
})
|
||||
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
outputs []*s3.ListObjectsV2Output
|
||||
want string
|
||||
}{
|
||||
{name: "empty", outputs: []*s3.ListObjectsV2Output{{IsTruncated: boolPtr(true)}}, want: "empty continuation token"},
|
||||
{name: "repeated", outputs: []*s3.ListObjectsV2Output{{IsTruncated: boolPtr(true), NextContinuationToken: strPtr("again")}, {IsTruncated: boolPtr(true), NextContinuationToken: strPtr("again")}}, want: "repeated continuation token"},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
_, err := (&S3Backend{bucket: "bucket-1", client: &fakeS3API{listOutputs: test.outputs}}).List(context.Background(), "prefix/")
|
||||
if err == nil || !strings.Contains(err.Error(), test.want) || !strings.Contains(err.Error(), "bucket-1") || !strings.Contains(err.Error(), "prefix/") {
|
||||
t.Fatalf("List() error = %v, want contextual %q", err, test.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func (f *fakeS3API) GetObject(_ context.Context, params *s3.GetObjectInput, _ ...func(*s3.Options)) (*s3.GetObjectOutput, error) {
|
||||
f.lastGet = params
|
||||
if f.getErr != nil {
|
||||
@@ -54,7 +102,7 @@ func (f *fakeS3API) GetObject(_ context.Context, params *s3.GetObjectInput, _ ..
|
||||
if body == nil {
|
||||
body = io.NopCloser(strings.NewReader(""))
|
||||
}
|
||||
return &s3.GetObjectOutput{Body: body}, nil
|
||||
return &s3.GetObjectOutput{Body: body, ContentLength: f.getSize, ETag: f.getETag, LastModified: f.getLastModified}, nil
|
||||
}
|
||||
|
||||
func (f *fakeS3API) PutObject(_ context.Context, params *s3.PutObjectInput, _ ...func(*s3.Options)) (*s3.PutObjectOutput, error) {
|
||||
@@ -126,6 +174,33 @@ func TestS3BackendDownloadCreatesParentDirectory(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestS3BackendReadReturnsOpenedObjectMetadata(t *testing.T) {
|
||||
lastModified := time.Date(2026, 8, 11, 1, 2, 3, 0, time.UTC)
|
||||
client := &fakeS3API{
|
||||
getBody: io.NopCloser(strings.NewReader("locks")),
|
||||
getSize: int64Ptr(5),
|
||||
getETag: strPtr(`"generation"`),
|
||||
getLastModified: &lastModified,
|
||||
}
|
||||
backend := &S3Backend{bucket: "bucket-1", client: client}
|
||||
|
||||
info, body, err := backend.Read(context.Background(), `sessions\locks.yml`)
|
||||
if err != nil {
|
||||
t.Fatalf("Read() error = %v", err)
|
||||
}
|
||||
data, readErr := io.ReadAll(body)
|
||||
closeErr := body.Close()
|
||||
if readErr != nil || closeErr != nil {
|
||||
t.Fatalf("read body error=%v close error=%v", readErr, closeErr)
|
||||
}
|
||||
if info.Key != "sessions/locks.yml" || info.Size != 5 || info.ETag != "generation" || info.LastModified == nil || !info.LastModified.Equal(lastModified) {
|
||||
t.Fatalf("Read() info = %#v, want opened object metadata", info)
|
||||
}
|
||||
if string(data) != "locks" || client.lastGet == nil || *client.lastGet.Key != "sessions/locks.yml" {
|
||||
t.Fatalf("Read() data=%q request=%#v", data, client.lastGet)
|
||||
}
|
||||
}
|
||||
|
||||
func TestS3BackendUploadAndExists(t *testing.T) {
|
||||
client := &fakeS3API{putOut: &s3.PutObjectOutput{ETag: strPtr(`"etag123"`)}}
|
||||
backend := &S3Backend{bucket: "bucket-1", client: client}
|
||||
@@ -160,6 +235,22 @@ func TestS3BackendUploadAndExists(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestS3BackendConditionalUploadUsesProviderPrecondition(t *testing.T) {
|
||||
client := &fakeS3API{putOut: &s3.PutObjectOutput{ETag: strPtr(`"etag123"`)}}
|
||||
backend := &S3Backend{bucket: "bucket-1", client: client}
|
||||
if _, err := backend.UploadConditional(context.Background(), strings.NewReader("payload"), "locks.yml", UploadOptions{}, WriteCondition{MatchETag: "before"}); err != nil {
|
||||
t.Fatalf("UploadConditional() error = %v", err)
|
||||
}
|
||||
if client.lastPut == nil || client.lastPut.IfMatch == nil || *client.lastPut.IfMatch != "before" || client.lastPut.IfNoneMatch != nil {
|
||||
t.Fatalf("PutObject conditional input = %#v", client.lastPut)
|
||||
}
|
||||
client.putErr = &smithy.GenericAPIError{Code: "PreconditionFailed", Message: "changed"}
|
||||
_, err := backend.UploadConditional(context.Background(), strings.NewReader("payload"), "locks.yml", UploadOptions{}, WriteCondition{RequireAbsent: true})
|
||||
if !errors.Is(err, ErrConditionNotMet) {
|
||||
t.Fatalf("UploadConditional() error = %v, want ErrConditionNotMet", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestS3BackendUploadMissingLocalFile(t *testing.T) {
|
||||
backend := &S3Backend{bucket: "bucket-1", client: &fakeS3API{}}
|
||||
_, err := backend.Upload(context.Background(), filepath.Join(t.TempDir(), "missing.txt"), "key.txt", UploadOptions{})
|
||||
@@ -249,5 +340,6 @@ func TestNewS3BackendFromConfigFallsBackWhenCredentialEnvMissing(t *testing.T) {
|
||||
|
||||
func strPtr(v string) *string { return &v }
|
||||
func int64Ptr(v int64) *int64 { return &v }
|
||||
func boolPtr(v bool) *bool { return &v }
|
||||
|
||||
var _ s3API = (*fakeS3API)(nil)
|
||||
|
||||
391
internal/adapters/subprocess/diagnostics.go
Normal file
391
internal/adapters/subprocess/diagnostics.go
Normal file
@@ -0,0 +1,391 @@
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
const (
|
||||
// MaxStdoutDiagnosticBytes bounds persisted stdout from one external command.
|
||||
MaxStdoutDiagnosticBytes int64 = 8 * 1024 * 1024
|
||||
// MaxStderrDiagnosticBytes bounds persisted stderr from one external command.
|
||||
MaxStderrDiagnosticBytes int64 = 8 * 1024 * 1024
|
||||
diagnosticTailBytes = 2048
|
||||
)
|
||||
|
||||
var inheritedEnvironmentNames = map[string]struct{}{
|
||||
"COMSPEC": {},
|
||||
"HOME": {},
|
||||
"PATH": {},
|
||||
"SYSTEMROOT": {},
|
||||
"TMP": {},
|
||||
"TMPDIR": {},
|
||||
"TEMP": {},
|
||||
"WINDIR": {},
|
||||
// These test-only helper destinations let the adapter package tests exercise
|
||||
// real command invocation without widening the production environment.
|
||||
"AUDITA_HELPER_RECORD_PATH": {},
|
||||
"AUDITA_HELPER_MODE": {},
|
||||
"GO_WANT_AUDITA_HELPER": {},
|
||||
"GO_WANT_SCRIPTORIUM_HELPER": {},
|
||||
"GO_WANT_SERIATIM_HELPER": {},
|
||||
"GO_WANT_SUBPROCESS_HELPER": {},
|
||||
"NOTARIUS_CAPTURE_DIR": {},
|
||||
"NOTARIUS_RECEIPT_FIXTURE": {},
|
||||
"SCRIPTORIUM_HELPER_RECORD_PATH": {},
|
||||
"SCRIPTORIUM_HELPER_MODE": {},
|
||||
"SERIATIM_HELPER_RECORD_PATH": {},
|
||||
"SERIATIM_HELPER_MODE": {},
|
||||
}
|
||||
|
||||
var sensitiveEnvironmentNames = map[string]struct{}{
|
||||
"ANTHROPIC_API_KEY": {},
|
||||
"API_KEY": {},
|
||||
"AUDITA_LLM_API_KEY": {},
|
||||
"AWS_ACCESS_KEY_ID": {},
|
||||
"AWS_SECRET_ACCESS_KEY": {},
|
||||
"AWS_SESSION_TOKEN": {},
|
||||
"OPENAI_API_KEY": {},
|
||||
"OPENROUTER_API_KEY": {},
|
||||
}
|
||||
|
||||
type captureLimitError struct {
|
||||
stream string
|
||||
owner string
|
||||
limit int64
|
||||
}
|
||||
|
||||
func (e *captureLimitError) Error() string {
|
||||
return fmt.Sprintf("%s diagnostic capture for %s exceeded %d bytes", e.stream, e.owner, e.limit)
|
||||
}
|
||||
|
||||
type logWriters struct {
|
||||
files []*os.File
|
||||
Stdout io.Writer
|
||||
Stderr io.Writer
|
||||
|
||||
limits chan *captureLimitError
|
||||
mu sync.Mutex
|
||||
limit *captureLimitError
|
||||
stdout *diagnosticWriter
|
||||
stderr *diagnosticWriter
|
||||
}
|
||||
|
||||
type diagnosticWriter struct {
|
||||
logs *logWriters
|
||||
stream string
|
||||
owner string
|
||||
target io.Writer
|
||||
limit int64
|
||||
received int64
|
||||
persisted int64
|
||||
redactor streamRedactor
|
||||
tail []byte
|
||||
}
|
||||
|
||||
func openLogWriters(stdoutPath, stderrPath, owner string, sensitiveValues []string) (*logWriters, error) {
|
||||
logs := &logWriters{limits: make(chan *captureLimitError, 1)}
|
||||
cleanStdout := cleanLogPath(stdoutPath)
|
||||
cleanStderr := cleanLogPath(stderrPath)
|
||||
|
||||
stdoutFile, err := openDiagnosticFile(cleanStdout)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open stdout log: %w", err)
|
||||
}
|
||||
stderrFile := stdoutFile
|
||||
if cleanStdout != cleanStderr {
|
||||
stderrFile, err = openDiagnosticFile(cleanStderr)
|
||||
if err != nil {
|
||||
_ = stdoutFile.Close()
|
||||
return nil, fmt.Errorf("open stderr log: %w", err)
|
||||
}
|
||||
}
|
||||
if cleanStdout == cleanStderr {
|
||||
logs.files = []*os.File{stdoutFile}
|
||||
} else {
|
||||
logs.files = []*os.File{stdoutFile, stderrFile}
|
||||
}
|
||||
|
||||
logs.stdout = newDiagnosticWriter(logs, "stdout", owner, stdoutFile, MaxStdoutDiagnosticBytes, sensitiveValues)
|
||||
logs.stderr = newDiagnosticWriter(logs, "stderr", owner, stderrFile, MaxStderrDiagnosticBytes, sensitiveValues)
|
||||
logs.Stdout = logs.stdout
|
||||
logs.Stderr = logs.stderr
|
||||
return logs, nil
|
||||
}
|
||||
|
||||
func newDiagnosticWriter(logs *logWriters, stream, owner string, target io.Writer, limit int64, sensitiveValues []string) *diagnosticWriter {
|
||||
return &diagnosticWriter{
|
||||
logs: logs,
|
||||
stream: stream,
|
||||
owner: owner,
|
||||
target: target,
|
||||
limit: limit,
|
||||
redactor: newStreamRedactor(sensitiveValues),
|
||||
}
|
||||
}
|
||||
|
||||
func (w *diagnosticWriter) Write(data []byte) (int, error) {
|
||||
if w.received >= w.limit {
|
||||
return len(data), w.reachLimit()
|
||||
}
|
||||
accepted := data
|
||||
if remaining := w.limit - w.received; int64(len(accepted)) > remaining {
|
||||
accepted = accepted[:remaining]
|
||||
}
|
||||
w.received += int64(len(accepted))
|
||||
if err := w.writeRedacted(w.redactor.Write(accepted)); err != nil {
|
||||
return len(data), err
|
||||
}
|
||||
if len(accepted) != len(data) {
|
||||
return len(data), w.reachLimit()
|
||||
}
|
||||
return len(data), nil
|
||||
}
|
||||
|
||||
func (w *diagnosticWriter) Flush() error {
|
||||
return w.writeRedacted(w.redactor.Flush())
|
||||
}
|
||||
|
||||
func (w *diagnosticWriter) writeRedacted(data []byte) error {
|
||||
if len(data) == 0 {
|
||||
return nil
|
||||
}
|
||||
w.logs.mu.Lock()
|
||||
remaining := w.limit - w.persisted
|
||||
if remaining <= 0 {
|
||||
w.logs.mu.Unlock()
|
||||
return w.reachLimit()
|
||||
}
|
||||
toWrite := data
|
||||
exceeded := int64(len(data)) > remaining
|
||||
if exceeded {
|
||||
toWrite = toWrite[:remaining]
|
||||
}
|
||||
written, err := w.target.Write(toWrite)
|
||||
w.persisted += int64(written)
|
||||
w.retainTail(toWrite[:written])
|
||||
w.logs.mu.Unlock()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if exceeded {
|
||||
return w.reachLimit()
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (w *diagnosticWriter) Tail() string {
|
||||
w.logs.mu.Lock()
|
||||
defer w.logs.mu.Unlock()
|
||||
return strings.TrimSpace(string(w.tail))
|
||||
}
|
||||
|
||||
func (w *diagnosticWriter) retainTail(data []byte) {
|
||||
if len(data) >= diagnosticTailBytes {
|
||||
if cap(w.tail) < diagnosticTailBytes {
|
||||
w.tail = make([]byte, diagnosticTailBytes)
|
||||
} else {
|
||||
w.tail = w.tail[:diagnosticTailBytes]
|
||||
}
|
||||
copy(w.tail, data[len(data)-diagnosticTailBytes:])
|
||||
return
|
||||
}
|
||||
if cap(w.tail) < diagnosticTailBytes {
|
||||
retained := make([]byte, len(w.tail), diagnosticTailBytes)
|
||||
copy(retained, w.tail)
|
||||
w.tail = retained
|
||||
}
|
||||
if overflow := len(w.tail) + len(data) - diagnosticTailBytes; overflow > 0 {
|
||||
copy(w.tail, w.tail[overflow:])
|
||||
w.tail = w.tail[:len(w.tail)-overflow]
|
||||
}
|
||||
w.tail = append(w.tail, data...)
|
||||
}
|
||||
|
||||
func (w *diagnosticWriter) reachLimit() error {
|
||||
limit := &captureLimitError{stream: w.stream, owner: w.owner, limit: w.limit}
|
||||
w.logs.mu.Lock()
|
||||
if w.logs.limit == nil {
|
||||
w.logs.limit = limit
|
||||
w.logs.limits <- limit
|
||||
}
|
||||
w.logs.mu.Unlock()
|
||||
return limit
|
||||
}
|
||||
|
||||
func (l *logWriters) Limits() <-chan *captureLimitError { return l.limits }
|
||||
|
||||
func (l *logWriters) Limit() *captureLimitError {
|
||||
l.mu.Lock()
|
||||
defer l.mu.Unlock()
|
||||
return l.limit
|
||||
}
|
||||
|
||||
func (l *logWriters) Flush() error {
|
||||
return joinErrors(l.stdout.Flush(), l.stderr.Flush())
|
||||
}
|
||||
|
||||
func (l *logWriters) Close() {
|
||||
_ = l.Flush()
|
||||
for _, file := range l.files {
|
||||
_ = file.Close()
|
||||
}
|
||||
}
|
||||
|
||||
func cleanLogPath(path string) string {
|
||||
trimmed := strings.TrimSpace(path)
|
||||
if trimmed == "" {
|
||||
return ""
|
||||
}
|
||||
return filepath.Clean(trimmed)
|
||||
}
|
||||
|
||||
func openDiagnosticFile(path string) (*os.File, error) {
|
||||
if path == "" {
|
||||
return os.OpenFile(os.DevNull, os.O_WRONLY, 0)
|
||||
}
|
||||
if err := fileops.EnsureWorkspaceDirectory(filepath.Dir(path)); err != nil {
|
||||
return nil, fmt.Errorf("create log directory for %q: %w", path, err)
|
||||
}
|
||||
file, err := fileops.OpenFileConfined(path, os.O_WRONLY|os.O_CREATE|os.O_TRUNC, fileops.WorkspaceFileMode)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open log file %q: %w", path, err)
|
||||
}
|
||||
if err := file.Chmod(fileops.WorkspaceFileMode); err != nil {
|
||||
_ = file.Close()
|
||||
return nil, fmt.Errorf("set log file permissions %q: %w", path, err)
|
||||
}
|
||||
return file, nil
|
||||
}
|
||||
|
||||
func (r RunRequest) diagnosticOwner() string {
|
||||
if owner := strings.TrimSpace(r.DiagnosticOwner); owner != "" {
|
||||
return owner
|
||||
}
|
||||
return "subprocess"
|
||||
}
|
||||
|
||||
func buildChildEnvironment(base []string, overrides map[string]string) []string {
|
||||
values := make(map[string]string, len(inheritedEnvironmentNames)+len(overrides))
|
||||
for _, item := range base {
|
||||
name, value, ok := strings.Cut(item, "=")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
normalized := strings.ToUpper(name)
|
||||
if _, allowed := inheritedEnvironmentNames[normalized]; allowed {
|
||||
values[name] = value
|
||||
}
|
||||
}
|
||||
for name, value := range overrides {
|
||||
values[name] = value
|
||||
}
|
||||
names := make([]string, 0, len(values))
|
||||
for name := range values {
|
||||
names = append(names, name)
|
||||
}
|
||||
sort.Strings(names)
|
||||
out := make([]string, 0, len(names))
|
||||
for _, name := range names {
|
||||
out = append(out, name+"="+values[name])
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func sensitiveEnvironmentValues(environment []string, additionalNames []string) []string {
|
||||
names := make(map[string]struct{}, len(sensitiveEnvironmentNames)+len(additionalNames))
|
||||
for name := range sensitiveEnvironmentNames {
|
||||
names[name] = struct{}{}
|
||||
}
|
||||
for _, name := range additionalNames {
|
||||
if trimmed := strings.ToUpper(strings.TrimSpace(name)); trimmed != "" {
|
||||
names[trimmed] = struct{}{}
|
||||
}
|
||||
}
|
||||
values := make([]string, 0, len(names))
|
||||
for _, item := range environment {
|
||||
name, value, ok := strings.Cut(item, "=")
|
||||
if !ok || strings.TrimSpace(value) == "" {
|
||||
continue
|
||||
}
|
||||
if _, sensitive := names[strings.ToUpper(name)]; sensitive {
|
||||
values = append(values, value)
|
||||
}
|
||||
}
|
||||
return values
|
||||
}
|
||||
|
||||
type streamRedactor struct {
|
||||
values []string
|
||||
buffer []byte
|
||||
maxLen int
|
||||
}
|
||||
|
||||
func newStreamRedactor(values []string) streamRedactor {
|
||||
unique := make(map[string]struct{}, len(values))
|
||||
for _, value := range values {
|
||||
if value != "" {
|
||||
unique[value] = struct{}{}
|
||||
}
|
||||
}
|
||||
sorted := make([]string, 0, len(unique))
|
||||
for value := range unique {
|
||||
sorted = append(sorted, value)
|
||||
}
|
||||
sort.Slice(sorted, func(i, j int) bool { return len(sorted[i]) > len(sorted[j]) })
|
||||
maxLen := 1
|
||||
for _, value := range sorted {
|
||||
if len(value) > maxLen {
|
||||
maxLen = len(value)
|
||||
}
|
||||
}
|
||||
return streamRedactor{values: sorted, maxLen: maxLen}
|
||||
}
|
||||
|
||||
func (r *streamRedactor) Write(data []byte) []byte {
|
||||
r.buffer = append(r.buffer, data...)
|
||||
safeCut := len(r.buffer) - r.maxLen + 1
|
||||
if safeCut <= 0 {
|
||||
return nil
|
||||
}
|
||||
emitCut := safeCut
|
||||
for _, value := range r.values {
|
||||
start := 0
|
||||
for {
|
||||
index := bytes.Index(r.buffer[start:], []byte(value))
|
||||
if index < 0 {
|
||||
break
|
||||
}
|
||||
index += start
|
||||
if index+len(value) > safeCut && index < emitCut {
|
||||
emitCut = index
|
||||
}
|
||||
start = index + 1
|
||||
}
|
||||
}
|
||||
output := redactBytes(r.buffer[:emitCut], r.values)
|
||||
r.buffer = append(r.buffer[:0], r.buffer[emitCut:]...)
|
||||
return output
|
||||
}
|
||||
|
||||
func (r *streamRedactor) Flush() []byte {
|
||||
output := redactBytes(r.buffer, r.values)
|
||||
r.buffer = nil
|
||||
return output
|
||||
}
|
||||
|
||||
func redactBytes(data []byte, values []string) []byte {
|
||||
out := append([]byte(nil), data...)
|
||||
for _, value := range values {
|
||||
out = bytes.ReplaceAll(out, []byte(value), []byte("<redacted>"))
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -1,3 +1,2 @@
|
||||
// Package subprocess provides reusable process execution and generated-config helpers.
|
||||
package subprocess
|
||||
|
||||
|
||||
66
internal/adapters/subprocess/process_tree.go
Normal file
66
internal/adapters/subprocess/process_tree.go
Normal file
@@ -0,0 +1,66 @@
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"time"
|
||||
)
|
||||
|
||||
const (
|
||||
gracefulTerminationWait = 2 * time.Second
|
||||
forcefulTerminationWait = 2 * time.Second
|
||||
)
|
||||
|
||||
// ownedProcessTree owns every process started by a command invocation.
|
||||
// Implementations must tolerate a leader that has already exited.
|
||||
type ownedProcessTree interface {
|
||||
Start(*exec.Cmd) error
|
||||
TerminateGracefully() error
|
||||
TerminateForcefully() error
|
||||
Dispose() error
|
||||
}
|
||||
|
||||
func waitForOwnedCommand(ctx context.Context, tree ownedProcessTree, waitCh <-chan error, captureLimits <-chan *captureLimitError) (waitErr, ctxErr error, captureLimit *captureLimitError, cleanupErr error) {
|
||||
select {
|
||||
case waitErr = <-waitCh:
|
||||
return waitErr, nil, nil, nil
|
||||
case <-ctx.Done():
|
||||
ctxErr = ctx.Err()
|
||||
case captureLimit = <-captureLimits:
|
||||
}
|
||||
|
||||
cleanupErr = tree.TerminateGracefully()
|
||||
gracefulTimer := time.NewTimer(gracefulTerminationWait)
|
||||
defer gracefulTimer.Stop()
|
||||
|
||||
select {
|
||||
case waitErr = <-waitCh:
|
||||
// The leader may exit before descendants finish graceful shutdown.
|
||||
cleanupErr = joinErrors(cleanupErr, tree.TerminateForcefully())
|
||||
return waitErr, ctxErr, captureLimit, cleanupErr
|
||||
case <-gracefulTimer.C:
|
||||
}
|
||||
|
||||
cleanupErr = joinErrors(cleanupErr, tree.TerminateForcefully())
|
||||
forcefulTimer := time.NewTimer(forcefulTerminationWait)
|
||||
defer forcefulTimer.Stop()
|
||||
|
||||
select {
|
||||
case waitErr = <-waitCh:
|
||||
return waitErr, ctxErr, captureLimit, cleanupErr
|
||||
case <-forcefulTimer.C:
|
||||
return nil, ctxErr, captureLimit, joinErrors(cleanupErr, fmt.Errorf("owned subprocess did not reap within %s after forceful termination", forcefulTerminationWait))
|
||||
}
|
||||
}
|
||||
|
||||
func joinErrors(errs ...error) error {
|
||||
filtered := make([]error, 0, len(errs))
|
||||
for _, err := range errs {
|
||||
if err != nil {
|
||||
filtered = append(filtered, err)
|
||||
}
|
||||
}
|
||||
return errors.Join(filtered...)
|
||||
}
|
||||
227
internal/adapters/subprocess/process_tree_supported_test.go
Normal file
227
internal/adapters/subprocess/process_tree_supported_test.go
Normal file
@@ -0,0 +1,227 @@
|
||||
//go:build linux || darwin || windows
|
||||
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestRunCancellationTerminatesProcessTree(t *testing.T) {
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
resultCh := make(chan runOutcome, 1)
|
||||
req, sentinelPath := processTreeRequest(t)
|
||||
go func() {
|
||||
result, err := Run(ctx, req)
|
||||
resultCh <- runOutcome{result: result, err: err}
|
||||
}()
|
||||
|
||||
awaitHelperReady(t, req.EnvOverrides["SUBPROCESS_HELPER_READY_PATH"])
|
||||
cancel()
|
||||
|
||||
outcome := awaitRunOutcome(t, resultCh)
|
||||
if !outcome.result.Canceled {
|
||||
t.Fatalf("Canceled = %v, want true", outcome.result.Canceled)
|
||||
}
|
||||
if !errors.Is(outcome.err, context.Canceled) {
|
||||
t.Fatalf("error = %v, want context cancellation", outcome.err)
|
||||
}
|
||||
assertDescendantDidNotSurvive(t, sentinelPath)
|
||||
}
|
||||
|
||||
func TestRunTimeoutTerminatesProcessTree(t *testing.T) {
|
||||
req, sentinelPath := processTreeRequest(t)
|
||||
req.Timeout = 100 * time.Millisecond
|
||||
|
||||
result, err := Run(context.Background(), req)
|
||||
if !result.TimedOut {
|
||||
t.Fatalf("TimedOut = %v, want true", result.TimedOut)
|
||||
}
|
||||
if !errors.Is(err, context.DeadlineExceeded) {
|
||||
t.Fatalf("error = %v, want context deadline exceeded", err)
|
||||
}
|
||||
assertDescendantDidNotSurvive(t, sentinelPath)
|
||||
}
|
||||
|
||||
func TestRunCaptureLimitTerminatesProcessTree(t *testing.T) {
|
||||
req, sentinelPath := processTreeRequest(t)
|
||||
req.Args[len(req.Args)-1] = "tree-spam"
|
||||
|
||||
result, err := Run(context.Background(), req)
|
||||
if err == nil {
|
||||
t.Fatal("Run() error = nil, want capture-limit error")
|
||||
}
|
||||
if result.ExitCode == 0 {
|
||||
t.Fatalf("ExitCode = %d, want terminated process", result.ExitCode)
|
||||
}
|
||||
if !strings.Contains(err.Error(), "stdout diagnostic capture for subprocess exceeded") {
|
||||
t.Fatalf("error = %q, want stdout capture-limit context", err)
|
||||
}
|
||||
info, statErr := os.Stat(req.StdoutLogPath)
|
||||
if statErr != nil {
|
||||
t.Fatalf("stat stdout diagnostic: %v", statErr)
|
||||
}
|
||||
if info.Size() != MaxStdoutDiagnosticBytes {
|
||||
t.Fatalf("stdout diagnostic size = %d, want %d", info.Size(), MaxStdoutDiagnosticBytes)
|
||||
}
|
||||
assertDescendantDidNotSurvive(t, sentinelPath)
|
||||
}
|
||||
|
||||
func TestRunDisposesDescendantsAfterLeaderExit(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
mode string
|
||||
wantExitCode int
|
||||
wantWaitDelay bool
|
||||
ignoreTerm bool
|
||||
}{
|
||||
{name: "success retaining streams", mode: "leader-exit-retained", wantExitCode: 0, wantWaitDelay: true},
|
||||
{name: "success redirecting streams", mode: "leader-exit-redirected", wantExitCode: 0},
|
||||
{name: "failed leader", mode: "leader-fail-redirected", wantExitCode: 9, ignoreTerm: true},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
req, sentinelPath, releasePath := leaderExitRequest(t, tt.mode)
|
||||
if tt.ignoreTerm {
|
||||
req.EnvOverrides["SUBPROCESS_HELPER_IGNORE_TERM"] = "1"
|
||||
}
|
||||
outcomes := make(chan runOutcome, 1)
|
||||
go func() {
|
||||
result, err := Run(context.Background(), req)
|
||||
outcomes <- runOutcome{result: result, err: err}
|
||||
}()
|
||||
|
||||
var outcome runOutcome
|
||||
select {
|
||||
case outcome = <-outcomes:
|
||||
case <-time.After(6 * time.Second):
|
||||
t.Fatal("Run() did not complete bounded owned-tree disposal")
|
||||
}
|
||||
if outcome.result.ExitCode != tt.wantExitCode {
|
||||
t.Fatalf("ExitCode = %d, want %d", outcome.result.ExitCode, tt.wantExitCode)
|
||||
}
|
||||
if tt.wantWaitDelay {
|
||||
if !errors.Is(outcome.err, exec.ErrWaitDelay) {
|
||||
t.Fatalf("error = %v, want exec.ErrWaitDelay", outcome.err)
|
||||
}
|
||||
} else if tt.wantExitCode == 0 && outcome.err != nil {
|
||||
t.Fatalf("Run() error = %v, want nil", outcome.err)
|
||||
} else if tt.wantExitCode != 0 {
|
||||
var exitErr *exec.ExitError
|
||||
if !errors.As(outcome.err, &exitErr) || exitErr.ExitCode() != tt.wantExitCode {
|
||||
t.Fatalf("error = %v, want exit code %d", outcome.err, tt.wantExitCode)
|
||||
}
|
||||
}
|
||||
if _, err := os.Stat(req.EnvOverrides["SUBPROCESS_HELPER_READY_PATH"]); err != nil {
|
||||
t.Fatalf("descendant readiness file: %v", err)
|
||||
}
|
||||
if err := os.WriteFile(releasePath, []byte("release"), 0o600); err != nil {
|
||||
t.Fatalf("WriteFile(release) error = %v", err)
|
||||
}
|
||||
assertDescendantDidNotSurvive(t, sentinelPath)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
type runOutcome struct {
|
||||
result RunResult
|
||||
err error
|
||||
}
|
||||
|
||||
func processTreeRequest(t *testing.T) (RunRequest, string) {
|
||||
t.Helper()
|
||||
|
||||
executable, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
dir := t.TempDir()
|
||||
readyPath := filepath.Join(dir, "ready")
|
||||
sentinelPath := filepath.Join(dir, "descendant-survived")
|
||||
return RunRequest{
|
||||
Executable: executable,
|
||||
Args: []string{"-test.run=^TestSubprocessHelper$", "--", "tree"},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"SUBPROCESS_HELPER_READY_PATH": readyPath,
|
||||
"SUBPROCESS_HELPER_SENTINEL_PATH": sentinelPath,
|
||||
},
|
||||
StdoutLogPath: filepath.Join(dir, "stdout.log"),
|
||||
StderrLogPath: filepath.Join(dir, "stderr.log"),
|
||||
}, sentinelPath
|
||||
}
|
||||
|
||||
func leaderExitRequest(t *testing.T, mode string) (RunRequest, string, string) {
|
||||
t.Helper()
|
||||
|
||||
executable, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
dir := t.TempDir()
|
||||
readyPath := filepath.Join(dir, "ready")
|
||||
releasePath := filepath.Join(dir, "release")
|
||||
sentinelPath := filepath.Join(dir, "descendant-survived")
|
||||
return RunRequest{
|
||||
Executable: executable,
|
||||
Args: []string{"-test.run=^TestSubprocessHelper$", "--", mode},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"SUBPROCESS_HELPER_READY_PATH": readyPath,
|
||||
"SUBPROCESS_HELPER_RELEASE_PATH": releasePath,
|
||||
"SUBPROCESS_HELPER_SENTINEL_PATH": sentinelPath,
|
||||
},
|
||||
StdoutLogPath: filepath.Join(dir, "stdout.log"),
|
||||
StderrLogPath: filepath.Join(dir, "stderr.log"),
|
||||
}, sentinelPath, releasePath
|
||||
}
|
||||
|
||||
func awaitHelperReady(t *testing.T, readyPath string) {
|
||||
t.Helper()
|
||||
|
||||
deadline := time.Now().Add(2 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
if _, err := os.Stat(readyPath); err == nil {
|
||||
return
|
||||
} else if !errors.Is(err, os.ErrNotExist) {
|
||||
t.Fatalf("stat helper readiness: %v", err)
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
t.Fatal("helper did not start its descendant")
|
||||
}
|
||||
|
||||
func awaitRunOutcome(t *testing.T, outcomes <-chan runOutcome) runOutcome {
|
||||
t.Helper()
|
||||
|
||||
select {
|
||||
case outcome := <-outcomes:
|
||||
if outcome.err == nil {
|
||||
t.Fatal("Run() error = nil, want cancellation error")
|
||||
}
|
||||
return outcome
|
||||
case <-time.After(3 * time.Second):
|
||||
t.Fatal("Run() did not return after cancellation")
|
||||
return runOutcome{}
|
||||
}
|
||||
}
|
||||
|
||||
func assertDescendantDidNotSurvive(t *testing.T, sentinelPath string) {
|
||||
t.Helper()
|
||||
|
||||
time.Sleep(700 * time.Millisecond)
|
||||
if _, err := os.Stat(sentinelPath); err == nil {
|
||||
t.Fatal("descendant survived cancellation and wrote its sentinel")
|
||||
} else if !errors.Is(err, os.ErrNotExist) {
|
||||
t.Fatalf("stat descendant sentinel: %v", err)
|
||||
}
|
||||
}
|
||||
104
internal/adapters/subprocess/process_tree_unix.go
Normal file
104
internal/adapters/subprocess/process_tree_unix.go
Normal file
@@ -0,0 +1,104 @@
|
||||
//go:build linux || darwin
|
||||
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"os/exec"
|
||||
"syscall"
|
||||
"time"
|
||||
)
|
||||
|
||||
const processGroupPollInterval = 10 * time.Millisecond
|
||||
|
||||
type unixProcessTree struct {
|
||||
processGroupID int
|
||||
}
|
||||
|
||||
func newOwnedProcessTree() (ownedProcessTree, error) {
|
||||
return &unixProcessTree{}, nil
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) Start(cmd *exec.Cmd) error {
|
||||
cmd.SysProcAttr = &syscall.SysProcAttr{Setpgid: true}
|
||||
if err := cmd.Start(); err != nil {
|
||||
return err
|
||||
}
|
||||
tree.processGroupID = cmd.Process.Pid
|
||||
return nil
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) TerminateGracefully() error {
|
||||
return tree.signal(syscall.SIGTERM)
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) TerminateForcefully() error {
|
||||
return tree.signal(syscall.SIGKILL)
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) Dispose() error {
|
||||
hasMembers, err := tree.hasMembers()
|
||||
if err != nil || !hasMembers {
|
||||
return err
|
||||
}
|
||||
|
||||
cleanupErr := tree.TerminateGracefully()
|
||||
empty, waitErr := tree.waitUntilEmpty(gracefulTerminationWait)
|
||||
cleanupErr = joinErrors(cleanupErr, waitErr)
|
||||
if empty {
|
||||
return cleanupErr
|
||||
}
|
||||
|
||||
cleanupErr = joinErrors(cleanupErr, tree.TerminateForcefully())
|
||||
empty, waitErr = tree.waitUntilEmpty(forcefulTerminationWait)
|
||||
cleanupErr = joinErrors(cleanupErr, waitErr)
|
||||
if !empty {
|
||||
cleanupErr = joinErrors(cleanupErr, fmt.Errorf("owned subprocess group did not exit within %s after forceful termination", forcefulTerminationWait))
|
||||
}
|
||||
return cleanupErr
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) signal(signal syscall.Signal) error {
|
||||
if tree.processGroupID <= 0 {
|
||||
return nil
|
||||
}
|
||||
err := syscall.Kill(-tree.processGroupID, signal)
|
||||
if errors.Is(err, syscall.ESRCH) || errors.Is(err, os.ErrProcessDone) {
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) hasMembers() (bool, error) {
|
||||
if tree.processGroupID <= 0 {
|
||||
return false, nil
|
||||
}
|
||||
err := syscall.Kill(-tree.processGroupID, 0)
|
||||
if err == nil || errors.Is(err, syscall.EPERM) {
|
||||
return true, nil
|
||||
}
|
||||
if errors.Is(err, syscall.ESRCH) || errors.Is(err, os.ErrProcessDone) {
|
||||
return false, nil
|
||||
}
|
||||
return false, fmt.Errorf("inspect owned subprocess group: %w", err)
|
||||
}
|
||||
|
||||
func (tree *unixProcessTree) waitUntilEmpty(timeout time.Duration) (bool, error) {
|
||||
deadline := time.Now().Add(timeout)
|
||||
for {
|
||||
hasMembers, err := tree.hasMembers()
|
||||
if err != nil || !hasMembers {
|
||||
return !hasMembers, err
|
||||
}
|
||||
remaining := time.Until(deadline)
|
||||
if remaining <= 0 {
|
||||
return false, nil
|
||||
}
|
||||
if remaining > processGroupPollInterval {
|
||||
remaining = processGroupPollInterval
|
||||
}
|
||||
time.Sleep(remaining)
|
||||
}
|
||||
}
|
||||
12
internal/adapters/subprocess/process_tree_unsupported.go
Normal file
12
internal/adapters/subprocess/process_tree_unsupported.go
Normal file
@@ -0,0 +1,12 @@
|
||||
//go:build !linux && !darwin && !windows
|
||||
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"runtime"
|
||||
)
|
||||
|
||||
func newOwnedProcessTree() (ownedProcessTree, error) {
|
||||
return nil, fmt.Errorf("owned subprocess trees are unsupported on %s", runtime.GOOS)
|
||||
}
|
||||
122
internal/adapters/subprocess/process_tree_windows.go
Normal file
122
internal/adapters/subprocess/process_tree_windows.go
Normal file
@@ -0,0 +1,122 @@
|
||||
//go:build windows
|
||||
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os/exec"
|
||||
"syscall"
|
||||
"unsafe"
|
||||
|
||||
"golang.org/x/sys/windows"
|
||||
)
|
||||
|
||||
type windowsProcessTree struct {
|
||||
job windows.Handle
|
||||
}
|
||||
|
||||
func newOwnedProcessTree() (ownedProcessTree, error) {
|
||||
return &windowsProcessTree{}, nil
|
||||
}
|
||||
|
||||
func (tree *windowsProcessTree) Start(cmd *exec.Cmd) error {
|
||||
job, err := windows.CreateJobObject(nil, nil)
|
||||
if err != nil {
|
||||
return fmt.Errorf("create job object: %w", err)
|
||||
}
|
||||
|
||||
limits := windows.JOBOBJECT_EXTENDED_LIMIT_INFORMATION{}
|
||||
limits.BasicLimitInformation.LimitFlags = windows.JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE
|
||||
if _, err := windows.SetInformationJobObject(job, windows.JobObjectExtendedLimitInformation, uintptr(unsafe.Pointer(&limits)), uint32(unsafe.Sizeof(limits))); err != nil {
|
||||
_ = windows.CloseHandle(job)
|
||||
return fmt.Errorf("configure job object: %w", err)
|
||||
}
|
||||
|
||||
cmd.SysProcAttr = &syscall.SysProcAttr{CreationFlags: windows.CREATE_SUSPENDED}
|
||||
if err := cmd.Start(); err != nil {
|
||||
_ = windows.CloseHandle(job)
|
||||
return err
|
||||
}
|
||||
|
||||
process, err := windows.OpenProcess(windows.PROCESS_SET_QUOTA|windows.PROCESS_TERMINATE, false, uint32(cmd.Process.Pid))
|
||||
if err == nil {
|
||||
err = windows.AssignProcessToJobObject(job, process)
|
||||
_ = windows.CloseHandle(process)
|
||||
}
|
||||
if err == nil {
|
||||
err = resumeInitialThread(uint32(cmd.Process.Pid))
|
||||
}
|
||||
if err != nil {
|
||||
killErr := cmd.Process.Kill()
|
||||
waitErr := cmd.Wait()
|
||||
_ = windows.CloseHandle(job)
|
||||
return joinErrors(fmt.Errorf("assign process to job object: %w", err), killErr, waitErr)
|
||||
}
|
||||
|
||||
tree.job = job
|
||||
return nil
|
||||
}
|
||||
|
||||
func resumeInitialThread(processID uint32) error {
|
||||
snapshot, err := windows.CreateToolhelp32Snapshot(windows.TH32CS_SNAPTHREAD, 0)
|
||||
if err != nil {
|
||||
return fmt.Errorf("snapshot initial thread: %w", err)
|
||||
}
|
||||
defer func() { _ = windows.CloseHandle(snapshot) }()
|
||||
|
||||
entry := windows.ThreadEntry32{Size: uint32(unsafe.Sizeof(windows.ThreadEntry32{}))}
|
||||
if err := windows.Thread32First(snapshot, &entry); err != nil {
|
||||
return fmt.Errorf("find initial thread: %w", err)
|
||||
}
|
||||
for {
|
||||
if entry.OwnerProcessID != processID {
|
||||
// Keep enumerating until the suspended process's only initial thread
|
||||
// is found.
|
||||
} else {
|
||||
thread, openErr := windows.OpenThread(windows.THREAD_SUSPEND_RESUME, false, entry.ThreadID)
|
||||
if openErr != nil {
|
||||
return fmt.Errorf("open initial thread: %w", openErr)
|
||||
}
|
||||
defer func() { _ = windows.CloseHandle(thread) }()
|
||||
if _, resumeErr := windows.ResumeThread(thread); resumeErr != nil {
|
||||
return fmt.Errorf("resume initial thread: %w", resumeErr)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
if err := windows.Thread32Next(snapshot, &entry); err != nil {
|
||||
if errors.Is(err, windows.ERROR_NO_MORE_FILES) {
|
||||
break
|
||||
}
|
||||
return fmt.Errorf("find initial thread: %w", err)
|
||||
}
|
||||
}
|
||||
return fmt.Errorf("find initial thread: no thread found for process %d", processID)
|
||||
}
|
||||
|
||||
func (tree *windowsProcessTree) TerminateGracefully() error {
|
||||
// Windows jobs have no portable graceful signal. Terminating the owned job
|
||||
// is the safe fallback and prevents a descendant from escaping cleanup.
|
||||
return tree.terminate()
|
||||
}
|
||||
|
||||
func (tree *windowsProcessTree) TerminateForcefully() error {
|
||||
return tree.terminate()
|
||||
}
|
||||
|
||||
func (tree *windowsProcessTree) Dispose() error {
|
||||
if tree.job == 0 {
|
||||
return nil
|
||||
}
|
||||
err := windows.CloseHandle(tree.job)
|
||||
tree.job = 0
|
||||
return err
|
||||
}
|
||||
|
||||
func (tree *windowsProcessTree) terminate() error {
|
||||
if tree.job == 0 {
|
||||
return nil
|
||||
}
|
||||
return windows.TerminateJobObject(tree.job, 1)
|
||||
}
|
||||
@@ -2,28 +2,27 @@ package subprocess
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
"gopkg.in/yaml.v3"
|
||||
)
|
||||
|
||||
// RunRequest defines a subprocess invocation.
|
||||
type RunRequest struct {
|
||||
Executable string
|
||||
Args []string
|
||||
WorkingDir string
|
||||
EnvOverrides map[string]string
|
||||
Timeout time.Duration
|
||||
StdoutLogPath string
|
||||
StderrLogPath string
|
||||
Executable string
|
||||
Args []string
|
||||
WorkingDir string
|
||||
EnvOverrides map[string]string
|
||||
SensitiveEnvNames []string
|
||||
DiagnosticOwner string
|
||||
Timeout time.Duration
|
||||
StdoutLogPath string
|
||||
StderrLogPath string
|
||||
}
|
||||
|
||||
// RunResult captures subprocess execution details.
|
||||
@@ -54,17 +53,26 @@ func Run(ctx context.Context, req RunRequest) (RunResult, error) {
|
||||
}
|
||||
defer cancel()
|
||||
|
||||
logs, err := openLogWriters(req.StdoutLogPath, req.StderrLogPath)
|
||||
childEnv := buildChildEnvironment(os.Environ(), req.EnvOverrides)
|
||||
logs, err := openLogWriters(req.StdoutLogPath, req.StderrLogPath, req.diagnosticOwner(), sensitiveEnvironmentValues(childEnv, req.SensitiveEnvNames))
|
||||
if err != nil {
|
||||
return RunResult{}, err
|
||||
}
|
||||
defer logs.Close()
|
||||
|
||||
cmd := exec.CommandContext(runCtx, req.Executable, req.Args...)
|
||||
tree, err := newOwnedProcessTree()
|
||||
if err != nil {
|
||||
return RunResult{}, fmt.Errorf("prepare owned subprocess tree: %w", err)
|
||||
}
|
||||
|
||||
cmd := exec.Command(req.Executable, req.Args...)
|
||||
cmd.Dir = req.WorkingDir
|
||||
cmd.Env = mergeEnv(os.Environ(), req.EnvOverrides)
|
||||
cmd.Env = childEnv
|
||||
cmd.Stdout = logs.Stdout
|
||||
cmd.Stderr = logs.Stderr
|
||||
// Streaming capture uses pipes. Bound their lifetime when a leader exits
|
||||
// while a descendant still holds a stream descriptor.
|
||||
cmd.WaitDelay = forcefulTerminationWait
|
||||
|
||||
started := time.Now().UTC()
|
||||
result := RunResult{
|
||||
@@ -74,45 +82,67 @@ func Run(ctx context.Context, req RunRequest) (RunResult, error) {
|
||||
StderrLogPath: req.StderrLogPath,
|
||||
}
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
if err := runCtx.Err(); err != nil {
|
||||
result.CompletedAt = time.Now().UTC()
|
||||
result.Duration = result.CompletedAt.Sub(result.StartedAt)
|
||||
return result, fmt.Errorf("command was not started: %w", err)
|
||||
}
|
||||
|
||||
if err := tree.Start(cmd); err != nil {
|
||||
result.CompletedAt = time.Now().UTC()
|
||||
result.Duration = result.CompletedAt.Sub(result.StartedAt)
|
||||
return result, fmt.Errorf("start command %q with args %v: %w", req.Executable, req.Args, err)
|
||||
}
|
||||
|
||||
waitErr := cmd.Wait()
|
||||
waitCh := make(chan error, 1)
|
||||
go func() { waitCh <- cmd.Wait() }()
|
||||
|
||||
waitErr, ctxErr, captureLimit, cleanupErr := waitForOwnedCommand(runCtx, tree, waitCh, logs.Limits())
|
||||
cleanupErr = joinErrors(cleanupErr, tree.Dispose())
|
||||
cleanupErr = joinErrors(cleanupErr, logs.Flush())
|
||||
if captureLimit == nil {
|
||||
captureLimit = logs.Limit()
|
||||
}
|
||||
result.CompletedAt = time.Now().UTC()
|
||||
result.Duration = result.CompletedAt.Sub(result.StartedAt)
|
||||
if cmd.ProcessState != nil {
|
||||
result.ExitCode = cmd.ProcessState.ExitCode()
|
||||
}
|
||||
|
||||
ctxErr := runCtx.Err()
|
||||
if errors.Is(ctxErr, context.DeadlineExceeded) {
|
||||
if ctxErr == context.DeadlineExceeded {
|
||||
result.TimedOut = true
|
||||
}
|
||||
if errors.Is(ctxErr, context.Canceled) && !result.TimedOut {
|
||||
if ctxErr == context.Canceled && !result.TimedOut {
|
||||
result.Canceled = true
|
||||
}
|
||||
|
||||
if waitErr == nil {
|
||||
if waitErr == nil && ctxErr == nil && cleanupErr == nil {
|
||||
return result, nil
|
||||
}
|
||||
|
||||
stderrTail := readRedactedTail(req.StderrLogPath, req.EnvOverrides, 2048)
|
||||
stderrTail := logs.stderr.Tail()
|
||||
diagnostics := buildDiagnostics(req, result, stderrTail)
|
||||
|
||||
if captureLimit != nil {
|
||||
if cause := joinErrors(waitErr, cleanupErr); cause != nil {
|
||||
return result, fmt.Errorf("%w (%s): %w", captureLimit, diagnostics, cause)
|
||||
}
|
||||
return result, fmt.Errorf("%w (%s)", captureLimit, diagnostics)
|
||||
}
|
||||
if result.TimedOut {
|
||||
return result, fmt.Errorf("command timed out after %s (%s)", req.Timeout, diagnostics)
|
||||
return result, fmt.Errorf("command timed out after %s (%s): %w", req.Timeout, diagnostics, joinErrors(ctxErr, waitErr, cleanupErr))
|
||||
}
|
||||
if result.Canceled {
|
||||
return result, fmt.Errorf("command canceled (%s)", diagnostics)
|
||||
return result, fmt.Errorf("command canceled (%s): %w", diagnostics, joinErrors(ctxErr, waitErr, cleanupErr))
|
||||
}
|
||||
if exitErr, ok := waitErr.(*exec.ExitError); ok {
|
||||
return result, fmt.Errorf("command failed with exit code %d (%s): %w", exitErr.ExitCode(), diagnostics, waitErr)
|
||||
return result, fmt.Errorf("command failed with exit code %d (%s): %w", exitErr.ExitCode(), diagnostics, joinErrors(waitErr, cleanupErr))
|
||||
}
|
||||
if cleanupErr != nil {
|
||||
return result, fmt.Errorf("command cleanup failed (%s): %w", diagnostics, joinErrors(waitErr, cleanupErr))
|
||||
}
|
||||
|
||||
return result, fmt.Errorf("command failed to run (%s): %w", diagnostics, waitErr)
|
||||
return result, fmt.Errorf("command failed to run (%s): %w", diagnostics, joinErrors(waitErr, cleanupErr))
|
||||
}
|
||||
|
||||
// WriteYAMLAtomic marshals value as YAML and atomically writes it to path.
|
||||
@@ -127,141 +157,17 @@ func WriteYAMLAtomic(path string, value any, perm os.FileMode) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// WriteFileAtomic writes bytes via same-directory temp file + atomic rename.
|
||||
// WriteFileAtomic writes bytes through the shared durable replacement primitive.
|
||||
func WriteFileAtomic(path string, data []byte, perm os.FileMode) error {
|
||||
if strings.TrimSpace(path) == "" {
|
||||
return fmt.Errorf("write file: path is required")
|
||||
}
|
||||
|
||||
dir := filepath.Dir(path)
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
return fmt.Errorf("create parent directory %q: %w", dir, err)
|
||||
if err := fileops.WriteFileAtomic(path, data, perm); err != nil {
|
||||
return fmt.Errorf("write file %q: %w", path, err)
|
||||
}
|
||||
|
||||
base := filepath.Base(path)
|
||||
tmp, err := os.CreateTemp(dir, "."+base+".tmp-*")
|
||||
if err != nil {
|
||||
return fmt.Errorf("create temp file: %w", err)
|
||||
}
|
||||
tmpPath := tmp.Name()
|
||||
removeTmp := true
|
||||
defer func() {
|
||||
if removeTmp {
|
||||
_ = os.Remove(tmpPath)
|
||||
}
|
||||
}()
|
||||
|
||||
if _, err := tmp.Write(data); err != nil {
|
||||
_ = tmp.Close()
|
||||
return fmt.Errorf("write temp file: %w", err)
|
||||
}
|
||||
if err := tmp.Sync(); err != nil {
|
||||
_ = tmp.Close()
|
||||
return fmt.Errorf("sync temp file: %w", err)
|
||||
}
|
||||
if err := tmp.Close(); err != nil {
|
||||
return fmt.Errorf("close temp file: %w", err)
|
||||
}
|
||||
if err := os.Chmod(tmpPath, perm); err != nil {
|
||||
return fmt.Errorf("chmod temp file: %w", err)
|
||||
}
|
||||
if err := os.Rename(tmpPath, path); err != nil {
|
||||
return fmt.Errorf("rename temp file: %w", err)
|
||||
}
|
||||
removeTmp = false
|
||||
return nil
|
||||
}
|
||||
|
||||
type logWriters struct {
|
||||
files []*os.File
|
||||
Stdout io.Writer
|
||||
Stderr io.Writer
|
||||
}
|
||||
|
||||
func (l *logWriters) Close() {
|
||||
for _, f := range l.files {
|
||||
_ = f.Close()
|
||||
}
|
||||
}
|
||||
|
||||
func openLogWriters(stdoutPath, stderrPath string) (*logWriters, error) {
|
||||
cleanStdout := cleanLogPath(stdoutPath)
|
||||
cleanStderr := cleanLogPath(stderrPath)
|
||||
|
||||
// Keep stdout/stderr on the same file descriptor when both paths target
|
||||
// the same file to avoid descriptor aliasing surprises across runtimes.
|
||||
if cleanStdout != "" && cleanStdout == cleanStderr {
|
||||
f, err := openLogFile(cleanStdout)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open shared stdout/stderr log %q: %w", cleanStdout, err)
|
||||
}
|
||||
return &logWriters{
|
||||
files: []*os.File{f},
|
||||
Stdout: f,
|
||||
Stderr: f,
|
||||
}, nil
|
||||
}
|
||||
|
||||
stdoutFile, stdoutWriter, err := logWriter(cleanStdout)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open stdout log: %w", err)
|
||||
}
|
||||
stderrFile, stderrWriter, err := logWriter(cleanStderr)
|
||||
if err != nil {
|
||||
closeFile(stdoutFile)
|
||||
return nil, fmt.Errorf("open stderr log: %w", err)
|
||||
}
|
||||
|
||||
files := make([]*os.File, 0, 2)
|
||||
if stdoutFile != nil {
|
||||
files = append(files, stdoutFile)
|
||||
}
|
||||
if stderrFile != nil {
|
||||
files = append(files, stderrFile)
|
||||
}
|
||||
return &logWriters{
|
||||
files: files,
|
||||
Stdout: stdoutWriter,
|
||||
Stderr: stderrWriter,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func cleanLogPath(path string) string {
|
||||
trimmed := strings.TrimSpace(path)
|
||||
if trimmed == "" {
|
||||
return ""
|
||||
}
|
||||
return filepath.Clean(trimmed)
|
||||
}
|
||||
|
||||
func logWriter(path string) (*os.File, io.Writer, error) {
|
||||
if strings.TrimSpace(path) == "" {
|
||||
return nil, io.Discard, nil
|
||||
}
|
||||
f, err := openLogFile(path)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
return f, f, nil
|
||||
}
|
||||
|
||||
func openLogFile(path string) (*os.File, error) {
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
return nil, fmt.Errorf("create log directory for %q: %w", path, err)
|
||||
}
|
||||
f, err := os.Create(path)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open log file %q: %w", path, err)
|
||||
}
|
||||
return f, nil
|
||||
}
|
||||
|
||||
func closeFile(f *os.File) {
|
||||
if f != nil {
|
||||
_ = f.Close()
|
||||
}
|
||||
}
|
||||
|
||||
func buildDiagnostics(req RunRequest, result RunResult, stderrTail string) string {
|
||||
details := fmt.Sprintf(
|
||||
"executable=%q args=%v cwd=%q timeout=%s exit_code=%d timed_out=%t canceled=%t stdout_log=%q stderr_log=%q",
|
||||
@@ -298,87 +204,3 @@ func fdDiagnosticsHint(exitCode int, stderrTail string) string {
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func readRedactedTail(path string, envOverrides map[string]string, maxBytes int64) string {
|
||||
if strings.TrimSpace(path) == "" || maxBytes <= 0 {
|
||||
return ""
|
||||
}
|
||||
f, err := os.Open(path)
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
info, err := f.Stat()
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
size := info.Size()
|
||||
start := int64(0)
|
||||
if size > maxBytes {
|
||||
start = size - maxBytes
|
||||
}
|
||||
if _, err := f.Seek(start, io.SeekStart); err != nil {
|
||||
return ""
|
||||
}
|
||||
data, err := io.ReadAll(f)
|
||||
if err != nil {
|
||||
return ""
|
||||
}
|
||||
tail := strings.TrimSpace(string(data))
|
||||
if tail == "" {
|
||||
return ""
|
||||
}
|
||||
return redactSensitiveTail(tail, envOverrides)
|
||||
}
|
||||
|
||||
func redactSensitiveTail(tail string, envOverrides map[string]string) string {
|
||||
out := tail
|
||||
for k, v := range envOverrides {
|
||||
if strings.TrimSpace(v) == "" {
|
||||
continue
|
||||
}
|
||||
if looksSensitiveEnvKey(k) {
|
||||
out = strings.ReplaceAll(out, v, "<redacted>")
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func looksSensitiveEnvKey(key string) bool {
|
||||
k := strings.ToUpper(strings.TrimSpace(key))
|
||||
return strings.Contains(k, "KEY") ||
|
||||
strings.Contains(k, "TOKEN") ||
|
||||
strings.Contains(k, "SECRET") ||
|
||||
strings.Contains(k, "PASSWORD")
|
||||
}
|
||||
|
||||
func mergeEnv(base []string, overrides map[string]string) []string {
|
||||
if len(overrides) == 0 {
|
||||
return base
|
||||
}
|
||||
|
||||
kv := make(map[string]string, len(base)+len(overrides))
|
||||
for _, item := range base {
|
||||
k, v, ok := strings.Cut(item, "=")
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
kv[k] = v
|
||||
}
|
||||
for k, v := range overrides {
|
||||
kv[k] = v
|
||||
}
|
||||
|
||||
keys := make([]string, 0, len(kv))
|
||||
for k := range kv {
|
||||
keys = append(keys, k)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
|
||||
out := make([]string, 0, len(keys))
|
||||
for _, k := range keys {
|
||||
out = append(out, k+"="+kv[k])
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
@@ -1,10 +1,17 @@
|
||||
package subprocess
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"os/exec"
|
||||
"os/signal"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"strconv"
|
||||
"strings"
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -120,6 +127,316 @@ func TestRunFailureRedactsSensitiveTail(t *testing.T) {
|
||||
if !strings.Contains(err.Error(), "<redacted>") {
|
||||
t.Fatalf("error = %q, want redacted stderr tail marker", err.Error())
|
||||
}
|
||||
for _, path := range []string{req.StdoutLogPath, req.StderrLogPath} {
|
||||
data, readErr := os.ReadFile(path)
|
||||
if readErr != nil {
|
||||
t.Fatalf("read diagnostic %q: %v", path, readErr)
|
||||
}
|
||||
if strings.Contains(string(data), secretValue) {
|
||||
t.Fatalf("diagnostic %q leaked secret: %q", path, data)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunRejectsSymlinkDiagnosticWithoutTruncatingTarget(t *testing.T) {
|
||||
if runtime.GOOS == "windows" {
|
||||
t.Skip("creating symlinks requires privileges that are not available on every Windows runner")
|
||||
}
|
||||
exe, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
|
||||
dir := t.TempDir()
|
||||
targetPath := filepath.Join(dir, "outside.log")
|
||||
const original = "must remain unchanged"
|
||||
if err := os.WriteFile(targetPath, []byte(original), 0o600); err != nil {
|
||||
t.Fatalf("WriteFile(target) error = %v", err)
|
||||
}
|
||||
stdoutPath := filepath.Join(dir, "stdout.log")
|
||||
if err := os.Symlink(targetPath, stdoutPath); err != nil {
|
||||
t.Fatalf("Symlink() error = %v", err)
|
||||
}
|
||||
|
||||
_, err = Run(context.Background(), RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=^TestSubprocessHelper$", "--", "success"},
|
||||
EnvOverrides: map[string]string{"GO_WANT_SUBPROCESS_HELPER": "1"},
|
||||
StdoutLogPath: stdoutPath,
|
||||
StderrLogPath: filepath.Join(dir, "stderr.log"),
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), "symbolic link") {
|
||||
t.Fatalf("Run() error = %v, want symbolic-link rejection", err)
|
||||
}
|
||||
data, readErr := os.ReadFile(targetPath)
|
||||
if readErr != nil {
|
||||
t.Fatalf("ReadFile(target) error = %v", readErr)
|
||||
}
|
||||
if string(data) != original {
|
||||
t.Fatalf("target content = %q, want %q", data, original)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunFailureUsesOpenedDiagnosticAfterPathReplacement(t *testing.T) {
|
||||
exe, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
|
||||
dir := t.TempDir()
|
||||
readyPath := filepath.Join(dir, "ready")
|
||||
releasePath := filepath.Join(dir, "release")
|
||||
stderrPath := filepath.Join(dir, "stderr.log")
|
||||
openedPath := filepath.Join(dir, "opened-stderr.log")
|
||||
const secretValue = "replacement-api-key-value"
|
||||
const commandContent = "trusted command failure"
|
||||
req := RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=^TestSubprocessHelper$", "--", "delayed-fail"},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"API_KEY": secretValue,
|
||||
"SUBPROCESS_HELPER_READY_PATH": readyPath,
|
||||
"SUBPROCESS_HELPER_RELEASE_PATH": releasePath,
|
||||
"SUBPROCESS_HELPER_STDERR": commandContent,
|
||||
},
|
||||
StdoutLogPath: filepath.Join(dir, "stdout.log"),
|
||||
StderrLogPath: stderrPath,
|
||||
}
|
||||
|
||||
resultCh := make(chan error, 1)
|
||||
go func() {
|
||||
_, runErr := Run(context.Background(), req)
|
||||
resultCh <- runErr
|
||||
}()
|
||||
waitForHelperFile(t, readyPath)
|
||||
if err := os.Rename(stderrPath, openedPath); err != nil {
|
||||
t.Fatalf("Rename(stderr log) error = %v", err)
|
||||
}
|
||||
if err := os.WriteFile(stderrPath, []byte(secretValue), 0o600); err != nil {
|
||||
t.Fatalf("WriteFile(replacement) error = %v", err)
|
||||
}
|
||||
if err := os.WriteFile(releasePath, []byte("continue"), 0o600); err != nil {
|
||||
t.Fatalf("WriteFile(release) error = %v", err)
|
||||
}
|
||||
|
||||
select {
|
||||
case runErr := <-resultCh:
|
||||
if runErr == nil {
|
||||
t.Fatal("Run() error = nil, want command failure")
|
||||
}
|
||||
if strings.Contains(runErr.Error(), secretValue) {
|
||||
t.Fatalf("error read replacement-path content: %q", runErr)
|
||||
}
|
||||
if !strings.Contains(runErr.Error(), commandContent) {
|
||||
t.Fatalf("error = %q, want retained command diagnostic", runErr)
|
||||
}
|
||||
case <-time.After(3 * time.Second):
|
||||
t.Fatal("Run() did not return after helper release")
|
||||
}
|
||||
|
||||
openedData, err := os.ReadFile(openedPath)
|
||||
if err != nil {
|
||||
t.Fatalf("ReadFile(opened diagnostic) error = %v", err)
|
||||
}
|
||||
if !strings.Contains(string(openedData), commandContent) {
|
||||
t.Fatalf("opened diagnostic = %q, want command content", openedData)
|
||||
}
|
||||
replacementData, err := os.ReadFile(stderrPath)
|
||||
if err != nil {
|
||||
t.Fatalf("ReadFile(replacement diagnostic) error = %v", err)
|
||||
}
|
||||
if string(replacementData) != secretValue {
|
||||
t.Fatalf("replacement diagnostic = %q, want %q", replacementData, secretValue)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunRedactsSplitCredentialInSeparateAndSharedDiagnostics(t *testing.T) {
|
||||
exe, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
const secretValue = "split-super-secret-value"
|
||||
|
||||
for _, shared := range []bool{false, true} {
|
||||
t.Run(map[bool]string{false: "separate", true: "shared"}[shared], func(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
stdoutPath := filepath.Join(dir, "stdout.log")
|
||||
stderrPath := filepath.Join(dir, "stderr.log")
|
||||
if shared {
|
||||
stderrPath = stdoutPath
|
||||
}
|
||||
req := RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=^TestSubprocessHelper$", "--", "splitsecret"},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"API_KEY": secretValue,
|
||||
},
|
||||
StdoutLogPath: stdoutPath,
|
||||
StderrLogPath: stderrPath,
|
||||
}
|
||||
|
||||
_, runErr := Run(context.Background(), req)
|
||||
if runErr == nil {
|
||||
t.Fatal("Run() error = nil, want command failure")
|
||||
}
|
||||
if strings.Contains(runErr.Error(), secretValue) || !strings.Contains(runErr.Error(), "<redacted>") {
|
||||
t.Fatalf("error = %q, want redacted credential", runErr)
|
||||
}
|
||||
paths := map[string]struct{}{stdoutPath: {}, stderrPath: {}}
|
||||
for path := range paths {
|
||||
data, readErr := os.ReadFile(path)
|
||||
if readErr != nil {
|
||||
t.Fatalf("ReadFile(%q) error = %v", path, readErr)
|
||||
}
|
||||
if strings.Contains(string(data), secretValue) || !strings.Contains(string(data), "<redacted>") {
|
||||
t.Fatalf("diagnostic %q = %q, want redacted credential", path, data)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunRedactsInheritedSensitiveEnvironment(t *testing.T) {
|
||||
exe, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
secretValue := "inherited-secret-value"
|
||||
t.Setenv("OPENROUTER_API_KEY", secretValue)
|
||||
dir := t.TempDir()
|
||||
req := RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=TestSubprocessHelper", "--", "echoenv"},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"SUBPROCESS_HELPER_ENV_KEY": "OPENROUTER_API_KEY",
|
||||
},
|
||||
StdoutLogPath: filepath.Join(dir, "stdout.log"),
|
||||
StderrLogPath: filepath.Join(dir, "stderr.log"),
|
||||
}
|
||||
_, err = Run(context.Background(), req)
|
||||
if err == nil {
|
||||
t.Fatal("Run() error = nil, want non-nil")
|
||||
}
|
||||
if strings.Contains(err.Error(), secretValue) {
|
||||
t.Fatalf("error leaked inherited secret: %q", err)
|
||||
}
|
||||
for _, path := range []string{req.StdoutLogPath, req.StderrLogPath} {
|
||||
data, readErr := os.ReadFile(path)
|
||||
if readErr != nil {
|
||||
t.Fatalf("read diagnostic %q: %v", path, readErr)
|
||||
}
|
||||
if strings.Contains(string(data), secretValue) {
|
||||
t.Fatalf("diagnostic %q leaked inherited secret: %q", path, data)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunRedactsSensitiveOutputAndErrorTail(t *testing.T) {
|
||||
exe, err := os.Executable()
|
||||
if err != nil {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
secretValue := "override-secret-value"
|
||||
dir := t.TempDir()
|
||||
req := RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=TestSubprocessHelper", "--", "echoenv"},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"SUBPROCESS_HELPER_ENV_KEY": "OPENROUTER_API_KEY",
|
||||
"OPENROUTER_API_KEY": secretValue,
|
||||
},
|
||||
StdoutLogPath: filepath.Join(dir, "stdout.log"),
|
||||
StderrLogPath: filepath.Join(dir, "stderr.log"),
|
||||
}
|
||||
_, err = Run(context.Background(), req)
|
||||
if err == nil {
|
||||
t.Fatal("Run() error = nil, want non-nil")
|
||||
}
|
||||
if strings.Contains(err.Error(), secretValue) || !strings.Contains(err.Error(), "<redacted>") {
|
||||
t.Fatalf("error = %q, want redacted secret", err)
|
||||
}
|
||||
for _, path := range []string{req.StdoutLogPath, req.StderrLogPath} {
|
||||
data, readErr := os.ReadFile(path)
|
||||
if readErr != nil {
|
||||
t.Fatalf("read diagnostic %q: %v", path, readErr)
|
||||
}
|
||||
if strings.Contains(string(data), secretValue) || !strings.Contains(string(data), "<redacted>") {
|
||||
t.Fatalf("diagnostic %q = %q, want redacted secret", path, data)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestStreamRedactorHandlesSplitAndOverlappingSecrets(t *testing.T) {
|
||||
redactor := newStreamRedactor([]string{"abc", "abcde", "cde", ""})
|
||||
var output bytes.Buffer
|
||||
output.Write(redactor.Write([]byte("start-ab")))
|
||||
output.Write(redactor.Write([]byte("cde-end")))
|
||||
output.Write(redactor.Flush())
|
||||
if got := output.String(); got != "start-<redacted>-end" {
|
||||
t.Fatalf("redacted output = %q, want one redacted marker", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDiagnosticWriterHonorsExactLimitAndCapPlusOne(t *testing.T) {
|
||||
exactLogs := &logWriters{limits: make(chan *captureLimitError, 1)}
|
||||
var exactOutput bytes.Buffer
|
||||
exact := newDiagnosticWriter(exactLogs, "stdout", "test", &exactOutput, 5, nil)
|
||||
if _, err := exact.Write([]byte("abcde")); err != nil {
|
||||
t.Fatalf("exact Write() error = %v", err)
|
||||
}
|
||||
if err := exact.Flush(); err != nil {
|
||||
t.Fatalf("exact Flush() error = %v", err)
|
||||
}
|
||||
if got := exactOutput.String(); got != "abcde" {
|
||||
t.Fatalf("exact output = %q, want abcde", got)
|
||||
}
|
||||
if exactLogs.Limit() != nil {
|
||||
t.Fatal("exact write recorded a capture limit")
|
||||
}
|
||||
|
||||
cappedLogs := &logWriters{limits: make(chan *captureLimitError, 1)}
|
||||
var cappedOutput bytes.Buffer
|
||||
capped := newDiagnosticWriter(cappedLogs, "stderr", "test", &cappedOutput, 5, nil)
|
||||
if _, err := capped.Write([]byte("abcdef")); err == nil {
|
||||
t.Fatal("cap-plus-one Write() error = nil, want capture limit")
|
||||
}
|
||||
if err := capped.Flush(); err != nil {
|
||||
t.Fatalf("cap-plus-one Flush() error = %v", err)
|
||||
}
|
||||
if got := cappedOutput.String(); got != "abcde" {
|
||||
t.Fatalf("capped output = %q, want abcde", got)
|
||||
}
|
||||
if limit := cappedLogs.Limit(); limit == nil || limit.stream != "stderr" || limit.limit != 5 {
|
||||
t.Fatalf("capture limit = %#v, want stderr limit 5", limit)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDiagnosticWriterRetainsBoundedRedactedTail(t *testing.T) {
|
||||
logs := &logWriters{limits: make(chan *captureLimitError, 1)}
|
||||
var output bytes.Buffer
|
||||
secret := "credential-value"
|
||||
writer := newDiagnosticWriter(logs, "stderr", "test", &output, 16*1024, []string{secret})
|
||||
prefix := strings.Repeat("x", diagnosticTailBytes+512)
|
||||
if _, err := writer.Write([]byte(prefix + secret[:7])); err != nil {
|
||||
t.Fatalf("first Write() error = %v", err)
|
||||
}
|
||||
if _, err := writer.Write([]byte(secret[7:] + "-failure")); err != nil {
|
||||
t.Fatalf("second Write() error = %v", err)
|
||||
}
|
||||
if err := writer.Flush(); err != nil {
|
||||
t.Fatalf("Flush() error = %v", err)
|
||||
}
|
||||
tail := writer.Tail()
|
||||
if len(tail) > diagnosticTailBytes {
|
||||
t.Fatalf("retained tail length = %d, want at most %d", len(tail), diagnosticTailBytes)
|
||||
}
|
||||
if strings.Contains(tail, secret) || !strings.Contains(tail, "<redacted>-failure") {
|
||||
t.Fatalf("retained tail = %q, want bounded redacted content", tail)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunFailureAddsBadDescriptorHint(t *testing.T) {
|
||||
@@ -184,15 +501,14 @@ func TestRunInheritsParentEnvironmentByDefault(t *testing.T) {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
|
||||
t.Setenv("GO_WANT_SUBPROCESS_HELPER", "1")
|
||||
t.Setenv("SUBPROCESS_HELPER_ENV_KEY", "SUBPROCESS_PARENT_VALUE")
|
||||
t.Setenv("SUBPROCESS_PARENT_VALUE", "inherited-value")
|
||||
t.Setenv("PATH", "inherited-value")
|
||||
|
||||
dir := t.TempDir()
|
||||
stdoutPath := filepath.Join(dir, "stdout.log")
|
||||
req := RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=TestSubprocessHelper", "--", "printenv"},
|
||||
EnvOverrides: map[string]string{"GO_WANT_SUBPROCESS_HELPER": "1", "SUBPROCESS_HELPER_ENV_KEY": "PATH"},
|
||||
StdoutLogPath: stdoutPath,
|
||||
}
|
||||
|
||||
@@ -214,16 +530,16 @@ func TestRunEnvOverridesWinOverInheritedValues(t *testing.T) {
|
||||
t.Fatalf("os.Executable() error = %v", err)
|
||||
}
|
||||
|
||||
t.Setenv("GO_WANT_SUBPROCESS_HELPER", "1")
|
||||
t.Setenv("SUBPROCESS_HELPER_ENV_KEY", "SUBPROCESS_PARENT_VALUE")
|
||||
t.Setenv("SUBPROCESS_PARENT_VALUE", "parent-value")
|
||||
|
||||
dir := t.TempDir()
|
||||
stdoutPath := filepath.Join(dir, "stdout.log")
|
||||
req := RunRequest{
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=TestSubprocessHelper", "--", "printenv"},
|
||||
EnvOverrides: map[string]string{"SUBPROCESS_PARENT_VALUE": "override-value"},
|
||||
Executable: exe,
|
||||
Args: []string{"-test.run=TestSubprocessHelper", "--", "printenv"},
|
||||
EnvOverrides: map[string]string{
|
||||
"GO_WANT_SUBPROCESS_HELPER": "1",
|
||||
"SUBPROCESS_HELPER_ENV_KEY": "SUBPROCESS_PARENT_VALUE",
|
||||
"SUBPROCESS_PARENT_VALUE": "override-value",
|
||||
},
|
||||
StdoutLogPath: stdoutPath,
|
||||
}
|
||||
|
||||
@@ -361,6 +677,30 @@ func TestSubprocessHelper(t *testing.T) {
|
||||
case "failbadfd":
|
||||
_, _ = os.Stderr.WriteString("OSError: [Errno 9] Bad file descriptor\n")
|
||||
os.Exit(120)
|
||||
case "delayed-fail":
|
||||
if err := os.WriteFile(os.Getenv("SUBPROCESS_HELPER_READY_PATH"), []byte("ready"), 0o600); err != nil {
|
||||
os.Exit(4)
|
||||
}
|
||||
deadline := time.Now().Add(2 * time.Second)
|
||||
for {
|
||||
if _, err := os.Stat(os.Getenv("SUBPROCESS_HELPER_RELEASE_PATH")); err == nil {
|
||||
break
|
||||
} else if !errors.Is(err, os.ErrNotExist) || time.Now().After(deadline) {
|
||||
os.Exit(5)
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
_, _ = os.Stderr.WriteString(os.Getenv("SUBPROCESS_HELPER_STDERR"))
|
||||
os.Exit(6)
|
||||
case "splitsecret":
|
||||
secret := os.Getenv("API_KEY")
|
||||
split := len(secret) / 2
|
||||
for _, stream := range []*os.File{os.Stdout, os.Stderr} {
|
||||
_, _ = stream.WriteString(secret[:split])
|
||||
time.Sleep(20 * time.Millisecond)
|
||||
_, _ = stream.WriteString(secret[split:] + "\n")
|
||||
}
|
||||
os.Exit(7)
|
||||
case "sleep":
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
os.Exit(0)
|
||||
@@ -368,7 +708,115 @@ func TestSubprocessHelper(t *testing.T) {
|
||||
key := os.Getenv("SUBPROCESS_HELPER_ENV_KEY")
|
||||
_, _ = os.Stdout.WriteString(os.Getenv(key) + "\n")
|
||||
os.Exit(0)
|
||||
case "echoenv":
|
||||
key := os.Getenv("SUBPROCESS_HELPER_ENV_KEY")
|
||||
value := os.Getenv(key)
|
||||
_, _ = os.Stdout.WriteString(value)
|
||||
_, _ = os.Stderr.WriteString(value)
|
||||
os.Exit(5)
|
||||
case "spam":
|
||||
chunk := strings.Repeat("x", 64*1024)
|
||||
count, _ := strconv.Atoi(os.Getenv("SUBPROCESS_HELPER_CHUNKS"))
|
||||
for range count {
|
||||
_, _ = os.Stdout.WriteString(chunk)
|
||||
}
|
||||
os.Exit(0)
|
||||
case "tree-spam":
|
||||
descendant := exec.Command(os.Args[0], "-test.run=^TestSubprocessHelper$", "--", "descendant")
|
||||
descendant.Env = append(os.Environ(), "GO_WANT_SUBPROCESS_HELPER=1")
|
||||
descendant.Stdout = os.Stdout
|
||||
descendant.Stderr = os.Stderr
|
||||
if err := descendant.Start(); err != nil {
|
||||
os.Exit(3)
|
||||
}
|
||||
if err := os.WriteFile(os.Getenv("SUBPROCESS_HELPER_READY_PATH"), []byte("ready"), 0o600); err != nil {
|
||||
os.Exit(4)
|
||||
}
|
||||
chunk := strings.Repeat("x", 64*1024)
|
||||
for {
|
||||
_, _ = os.Stdout.WriteString(chunk)
|
||||
}
|
||||
case "tree":
|
||||
descendant := exec.Command(os.Args[0], "-test.run=^TestSubprocessHelper$", "--", "descendant")
|
||||
descendant.Env = append(os.Environ(), "GO_WANT_SUBPROCESS_HELPER=1")
|
||||
descendant.Stdout = os.Stdout
|
||||
descendant.Stderr = os.Stderr
|
||||
if err := descendant.Start(); err != nil {
|
||||
os.Exit(3)
|
||||
}
|
||||
if err := os.WriteFile(os.Getenv("SUBPROCESS_HELPER_READY_PATH"), []byte("ready"), 0o600); err != nil {
|
||||
os.Exit(4)
|
||||
}
|
||||
time.Sleep(10 * time.Second)
|
||||
os.Exit(0)
|
||||
case "leader-exit-retained", "leader-exit-redirected", "leader-fail-redirected":
|
||||
descendant := exec.Command(os.Args[0], "-test.run=^TestSubprocessHelper$", "--", "descendant-after-release")
|
||||
descendant.Env = append(os.Environ(), "GO_WANT_SUBPROCESS_HELPER=1")
|
||||
if mode == "leader-exit-retained" {
|
||||
descendant.Stdout = os.Stdout
|
||||
descendant.Stderr = os.Stderr
|
||||
}
|
||||
if err := descendant.Start(); err != nil {
|
||||
os.Exit(3)
|
||||
}
|
||||
if !helperFileAppeared(os.Getenv("SUBPROCESS_HELPER_READY_PATH"), 2*time.Second) {
|
||||
os.Exit(4)
|
||||
}
|
||||
if mode == "leader-fail-redirected" {
|
||||
os.Exit(9)
|
||||
}
|
||||
os.Exit(0)
|
||||
case "descendant-after-release":
|
||||
if os.Getenv("SUBPROCESS_HELPER_IGNORE_TERM") == "1" {
|
||||
signal.Ignore(syscall.SIGTERM)
|
||||
}
|
||||
if err := os.WriteFile(os.Getenv("SUBPROCESS_HELPER_READY_PATH"), []byte("ready"), 0o600); err != nil {
|
||||
os.Exit(4)
|
||||
}
|
||||
deadline := time.Now().Add(10 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
if _, err := os.Stat(os.Getenv("SUBPROCESS_HELPER_RELEASE_PATH")); err == nil {
|
||||
_ = os.WriteFile(os.Getenv("SUBPROCESS_HELPER_SENTINEL_PATH"), []byte("survived"), 0o600)
|
||||
os.Exit(0)
|
||||
} else if !errors.Is(err, os.ErrNotExist) {
|
||||
os.Exit(5)
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
os.Exit(0)
|
||||
case "descendant":
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
_ = os.WriteFile(os.Getenv("SUBPROCESS_HELPER_SENTINEL_PATH"), []byte("survived"), 0o600)
|
||||
time.Sleep(10 * time.Second)
|
||||
os.Exit(0)
|
||||
default:
|
||||
os.Exit(2)
|
||||
}
|
||||
}
|
||||
|
||||
func waitForHelperFile(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(2 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
if _, err := os.Stat(path); err == nil {
|
||||
return
|
||||
} else if !errors.Is(err, os.ErrNotExist) {
|
||||
t.Fatalf("Stat(%q) error = %v", path, err)
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
t.Fatalf("helper file %q was not created", path)
|
||||
}
|
||||
|
||||
func helperFileAppeared(path string, timeout time.Duration) bool {
|
||||
deadline := time.Now().Add(timeout)
|
||||
for time.Now().Before(deadline) {
|
||||
if _, err := os.Stat(path); err == nil {
|
||||
return true
|
||||
} else if !errors.Is(err, os.ErrNotExist) {
|
||||
return false
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -2,8 +2,10 @@ package whisperx
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sync"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
var minimalTranscriptJSON = []byte(`{"schema":"speaker_transcript.v1","segments":[]}`)
|
||||
@@ -31,7 +33,8 @@ func (n *NoopClient) Transcribe(ctx context.Context, req TranscribeRequest) (Tra
|
||||
|
||||
// FakeClient captures requests and returns deterministic responses for tests.
|
||||
type FakeClient struct {
|
||||
Requests []TranscribeRequest
|
||||
requestsMu sync.RWMutex
|
||||
requests []TranscribeRequest
|
||||
Err error
|
||||
Result TranscribeResult
|
||||
TranscribeFn func(ctx context.Context, req TranscribeRequest) (TranscribeResult, error)
|
||||
@@ -42,7 +45,9 @@ func (f *FakeClient) Transcribe(ctx context.Context, req TranscribeRequest) (Tra
|
||||
if err := ctx.Err(); err != nil {
|
||||
return TranscribeResult{}, err
|
||||
}
|
||||
f.Requests = append(f.Requests, req)
|
||||
f.requestsMu.Lock()
|
||||
f.requests = append(f.requests, req)
|
||||
f.requestsMu.Unlock()
|
||||
if f.TranscribeFn != nil {
|
||||
return f.TranscribeFn(ctx, req)
|
||||
}
|
||||
@@ -65,12 +70,19 @@ func (f *FakeClient) Transcribe(ctx context.Context, req TranscribeRequest) (Tra
|
||||
return res, nil
|
||||
}
|
||||
|
||||
// RequestsSnapshot returns a copy of captured requests safe for concurrent test assertions.
|
||||
func (f *FakeClient) RequestsSnapshot() []TranscribeRequest {
|
||||
f.requestsMu.RLock()
|
||||
defer f.requestsMu.RUnlock()
|
||||
return append([]TranscribeRequest(nil), f.requests...)
|
||||
}
|
||||
|
||||
func writeMinimalJSON(path string) error {
|
||||
if path == "" {
|
||||
return nil
|
||||
}
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
if err := fileops.EnsureWorkspaceDirectory(filepath.Dir(path)); err != nil {
|
||||
return err
|
||||
}
|
||||
return os.WriteFile(path, minimalTranscriptJSON, 0o644)
|
||||
return fileops.WriteFileAtomic(path, minimalTranscriptJSON, fileops.WorkspaceFileMode)
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@ package whisperx
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"sync"
|
||||
"testing"
|
||||
)
|
||||
|
||||
@@ -14,8 +15,9 @@ func TestFakeClientCapturesRequestAndReturnsPath(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("Transcribe() error = %v", err)
|
||||
}
|
||||
if len(fake.Requests) != 1 || fake.Requests[0].SpeakerID != "alice" {
|
||||
t.Fatalf("requests = %#v, want one alice request", fake.Requests)
|
||||
requests := fake.RequestsSnapshot()
|
||||
if len(requests) != 1 || requests[0].SpeakerID != "alice" {
|
||||
t.Fatalf("requests = %#v, want one alice request", requests)
|
||||
}
|
||||
if res.OutputRawTranscriptPath != req.OutputRawTranscriptPath {
|
||||
t.Fatalf("output path = %q, want %q", res.OutputRawTranscriptPath, req.OutputRawTranscriptPath)
|
||||
@@ -29,3 +31,22 @@ func TestFakeClientError(t *testing.T) {
|
||||
t.Fatal("expected error, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestFakeClientRequestsSnapshotSupportsConcurrentCalls(t *testing.T) {
|
||||
fake := &FakeClient{}
|
||||
const callers = 16
|
||||
var group sync.WaitGroup
|
||||
group.Add(callers)
|
||||
for i := 0; i < callers; i++ {
|
||||
go func() {
|
||||
defer group.Done()
|
||||
if _, err := fake.Transcribe(context.Background(), TranscribeRequest{}); err != nil {
|
||||
t.Errorf("Transcribe() error = %v", err)
|
||||
}
|
||||
}()
|
||||
}
|
||||
group.Wait()
|
||||
if got := len(fake.RequestsSnapshot()); got != callers {
|
||||
t.Fatalf("captured requests = %d, want %d", got, callers)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
package whisperx
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
@@ -14,10 +13,16 @@ import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
const defaultMaxResponseBytes int64 = 10 * 1024 * 1024
|
||||
const (
|
||||
defaultMaxWhisperXResponseBytes int64 = 10 * 1024 * 1024
|
||||
whisperXUploadBufferSize = 32 * 1024
|
||||
)
|
||||
|
||||
// HTTPClientConfig contains parsed, deterministic WhisperX HTTP client settings.
|
||||
type HTTPClientConfig struct {
|
||||
@@ -39,6 +44,7 @@ type HTTPClient struct {
|
||||
retryDelay time.Duration
|
||||
httpClient *http.Client
|
||||
maxResponseBytes int64
|
||||
openAudio func(string) (io.ReadCloser, error)
|
||||
}
|
||||
|
||||
// NewHTTPClientFromConfigValues builds a client from config values and parses durations once.
|
||||
@@ -72,11 +78,11 @@ func NewHTTPClient(cfg HTTPClientConfig) (*HTTPClient, error) {
|
||||
return nil, fmt.Errorf("whisperx transcribe_url is required")
|
||||
}
|
||||
u, err := url.Parse(cfg.TranscribeURL)
|
||||
if err != nil || u.Scheme == "" || u.Host == "" {
|
||||
if err != nil || !u.IsAbs() || u.Host == "" || !isHTTPURLScheme(u.Scheme) {
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("invalid whisperx transcribe_url %q: %w", cfg.TranscribeURL, err)
|
||||
}
|
||||
return nil, fmt.Errorf("invalid whisperx transcribe_url %q", cfg.TranscribeURL)
|
||||
return nil, fmt.Errorf("invalid whisperx transcribe_url %q: must be an absolute http or https URL", cfg.TranscribeURL)
|
||||
}
|
||||
if cfg.Timeout <= 0 {
|
||||
return nil, fmt.Errorf("whisperx timeout must be > 0")
|
||||
@@ -93,12 +99,14 @@ func NewHTTPClient(cfg HTTPClientConfig) (*HTTPClient, error) {
|
||||
|
||||
client := cfg.HTTPClient
|
||||
if client == nil {
|
||||
client = &http.Client{}
|
||||
transport := http.DefaultTransport.(*http.Transport).Clone()
|
||||
transport.ExpectContinueTimeout = 100 * time.Millisecond
|
||||
client = &http.Client{Transport: transport}
|
||||
}
|
||||
|
||||
maxBytes := cfg.MaxResponseBytes
|
||||
if maxBytes <= 0 {
|
||||
maxBytes = defaultMaxResponseBytes
|
||||
maxBytes = defaultMaxWhisperXResponseBytes
|
||||
}
|
||||
|
||||
return &HTTPClient{
|
||||
@@ -109,6 +117,7 @@ func NewHTTPClient(cfg HTTPClientConfig) (*HTTPClient, error) {
|
||||
retryDelay: cfg.RetryDelay,
|
||||
httpClient: client,
|
||||
maxResponseBytes: maxBytes,
|
||||
openAudio: func(path string) (io.ReadCloser, error) { return os.Open(path) },
|
||||
}, nil
|
||||
}
|
||||
|
||||
@@ -147,7 +156,7 @@ func (c *HTTPClient) Transcribe(ctx context.Context, req TranscribeRequest) (Tra
|
||||
result.Duration = time.Since(start)
|
||||
return result, fmt.Errorf("whisperx attempt %d returned invalid json: %w", attempt, err)
|
||||
}
|
||||
if err := writeFileAtomic(req.OutputRawTranscriptPath, body, 0o644); err != nil {
|
||||
if err := writeFileAtomic(req.OutputRawTranscriptPath, body, fileops.WorkspaceFileMode); err != nil {
|
||||
result.Duration = time.Since(start)
|
||||
return result, fmt.Errorf("whisperx write transcript output %q: %w", req.OutputRawTranscriptPath, err)
|
||||
}
|
||||
@@ -183,52 +192,44 @@ func (c *HTTPClient) Transcribe(ctx context.Context, req TranscribeRequest) (Tra
|
||||
}
|
||||
|
||||
func (c *HTTPClient) doTranscribeAttempt(ctx context.Context, audioPath string) (int, []byte, error) {
|
||||
bodyBuf := &bytes.Buffer{}
|
||||
writer := multipart.NewWriter(bodyBuf)
|
||||
upload := newMultipartUpload(ctx, audioPath, c.language, c.openAudio)
|
||||
defer upload.Close()
|
||||
|
||||
fileWriter, err := writer.CreateFormFile("file", filepath.Base(audioPath))
|
||||
if err != nil {
|
||||
return 0, nil, fmt.Errorf("create multipart file field: %w", err)
|
||||
}
|
||||
|
||||
audioFile, err := os.Open(audioPath)
|
||||
if err != nil {
|
||||
return 0, nil, fmt.Errorf("open audio file %q: %w", audioPath, err)
|
||||
}
|
||||
if _, err := io.Copy(fileWriter, audioFile); err != nil {
|
||||
_ = audioFile.Close()
|
||||
return 0, nil, fmt.Errorf("copy audio file %q: %w", audioPath, err)
|
||||
}
|
||||
if err := audioFile.Close(); err != nil {
|
||||
return 0, nil, fmt.Errorf("close audio file %q: %w", audioPath, err)
|
||||
}
|
||||
|
||||
if err := writer.WriteField("language", c.language); err != nil {
|
||||
return 0, nil, fmt.Errorf("write language form field: %w", err)
|
||||
}
|
||||
if err := writer.Close(); err != nil {
|
||||
return 0, nil, fmt.Errorf("close multipart writer: %w", err)
|
||||
}
|
||||
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, c.url.String(), bodyBuf)
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, c.url.String(), upload)
|
||||
if err != nil {
|
||||
return 0, nil, fmt.Errorf("build whisperx request: %w", err)
|
||||
}
|
||||
req.Header.Set("Content-Type", writer.FormDataContentType())
|
||||
req.Header.Set("Content-Type", upload.contentType)
|
||||
req.Header.Set("Expect", "100-continue")
|
||||
|
||||
resp, err := c.httpClient.Do(req)
|
||||
if err != nil {
|
||||
_ = upload.Close()
|
||||
if producerErr := upload.Wait(); producerErr != nil {
|
||||
return 0, nil, fmt.Errorf("stream whisperx request body: %w", producerErr)
|
||||
}
|
||||
return 0, nil, fmt.Errorf("perform whisperx request: %w", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
data, err := readBounded(resp.Body, c.maxResponseBytes)
|
||||
if err != nil {
|
||||
return resp.StatusCode, nil, fmt.Errorf("read whisperx response body: %w", err)
|
||||
if resp.StatusCode < 200 || resp.StatusCode >= 300 {
|
||||
_ = upload.Close()
|
||||
if producerErr := upload.Wait(); producerErr != nil {
|
||||
return resp.StatusCode, nil, fmt.Errorf("stream whisperx request body: %w", producerErr)
|
||||
}
|
||||
if _, err := readWhisperXResponse(resp.Body, c.maxResponseBytes); err != nil {
|
||||
return resp.StatusCode, nil, fmt.Errorf("read whisperx response body: %w", err)
|
||||
}
|
||||
return resp.StatusCode, nil, fmt.Errorf("whisperx returned status %d", resp.StatusCode)
|
||||
}
|
||||
|
||||
if resp.StatusCode < 200 || resp.StatusCode >= 300 {
|
||||
return resp.StatusCode, nil, fmt.Errorf("whisperx returned status %d", resp.StatusCode)
|
||||
if err := upload.Wait(); err != nil {
|
||||
return resp.StatusCode, nil, fmt.Errorf("stream whisperx request body: %w", err)
|
||||
}
|
||||
|
||||
data, err := readWhisperXResponse(resp.Body, c.maxResponseBytes)
|
||||
if err != nil {
|
||||
return resp.StatusCode, nil, fmt.Errorf("read whisperx response body: %w", err)
|
||||
}
|
||||
return resp.StatusCode, data, nil
|
||||
}
|
||||
@@ -264,57 +265,188 @@ func (c *HTTPClient) shouldRetry(parent context.Context, err error, status int)
|
||||
return false
|
||||
}
|
||||
|
||||
func readBounded(r io.Reader, maxBytes int64) ([]byte, error) {
|
||||
func readWhisperXResponse(r io.Reader, maxBytes int64) ([]byte, error) {
|
||||
limited := io.LimitReader(r, maxBytes+1)
|
||||
data, err := io.ReadAll(limited)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if int64(len(data)) > maxBytes {
|
||||
return nil, fmt.Errorf("response exceeds max size %d bytes", maxBytes)
|
||||
return nil, fmt.Errorf("whisperx response exceeds configured limit of %d bytes", maxBytes)
|
||||
}
|
||||
return data, nil
|
||||
}
|
||||
|
||||
func isHTTPURLScheme(scheme string) bool {
|
||||
switch strings.ToLower(scheme) {
|
||||
case "http", "https":
|
||||
return true
|
||||
default:
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
type multipartUpload struct {
|
||||
reader *io.PipeReader
|
||||
writer *io.PipeWriter
|
||||
contentType string
|
||||
done chan struct{}
|
||||
|
||||
mu sync.Mutex
|
||||
audio io.Closer
|
||||
err error
|
||||
aborted bool
|
||||
}
|
||||
|
||||
func newMultipartUpload(ctx context.Context, audioPath, language string, openAudio func(string) (io.ReadCloser, error)) *multipartUpload {
|
||||
reader, writer := io.Pipe()
|
||||
multipartWriter := multipart.NewWriter(writer)
|
||||
upload := &multipartUpload{
|
||||
reader: reader,
|
||||
writer: writer,
|
||||
contentType: multipartWriter.FormDataContentType(),
|
||||
done: make(chan struct{}),
|
||||
}
|
||||
|
||||
go func() {
|
||||
err := upload.write(ctx, multipartWriter, audioPath, language, openAudio)
|
||||
if err != nil {
|
||||
_ = writer.CloseWithError(err)
|
||||
} else {
|
||||
_ = writer.Close()
|
||||
}
|
||||
upload.mu.Lock()
|
||||
upload.err = err
|
||||
upload.audio = nil
|
||||
upload.mu.Unlock()
|
||||
close(upload.done)
|
||||
}()
|
||||
|
||||
go func() {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
upload.abort()
|
||||
case <-upload.done:
|
||||
}
|
||||
}()
|
||||
|
||||
return upload
|
||||
}
|
||||
|
||||
func (u *multipartUpload) Read(p []byte) (int, error) {
|
||||
return u.reader.Read(p)
|
||||
}
|
||||
|
||||
func (u *multipartUpload) Close() error {
|
||||
u.abort()
|
||||
return nil
|
||||
}
|
||||
|
||||
func (u *multipartUpload) Wait() error {
|
||||
<-u.done
|
||||
u.mu.Lock()
|
||||
defer u.mu.Unlock()
|
||||
return u.err
|
||||
}
|
||||
|
||||
func (u *multipartUpload) write(ctx context.Context, writer *multipart.Writer, audioPath, language string, openAudio func(string) (io.ReadCloser, error)) error {
|
||||
fileWriter, err := writer.CreateFormFile("file", filepath.Base(audioPath))
|
||||
if err != nil {
|
||||
return u.producerError(ctx, fmt.Errorf("create multipart file field: %w", err))
|
||||
}
|
||||
|
||||
audioFile, err := openAudio(audioPath)
|
||||
if err != nil {
|
||||
return u.producerError(ctx, fmt.Errorf("open audio file %q: %w", audioPath, err))
|
||||
}
|
||||
u.setAudio(audioFile)
|
||||
|
||||
_, copyErr := io.CopyBuffer(fileWriter, &contextReader{ctx: ctx, reader: audioFile}, make([]byte, whisperXUploadBufferSize))
|
||||
closeErr := audioFile.Close()
|
||||
u.clearAudio(audioFile)
|
||||
if copyErr != nil {
|
||||
return u.producerError(ctx, fmt.Errorf("copy audio file %q: %w", audioPath, copyErr))
|
||||
}
|
||||
if closeErr != nil {
|
||||
return u.producerError(ctx, fmt.Errorf("close audio file %q: %w", audioPath, closeErr))
|
||||
}
|
||||
|
||||
if err := writer.WriteField("language", language); err != nil {
|
||||
return u.producerError(ctx, fmt.Errorf("write language form field: %w", err))
|
||||
}
|
||||
if err := writer.Close(); err != nil {
|
||||
return u.producerError(ctx, fmt.Errorf("close multipart writer: %w", err))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (u *multipartUpload) producerError(ctx context.Context, err error) error {
|
||||
if ctx.Err() != nil {
|
||||
return ctx.Err()
|
||||
}
|
||||
u.mu.Lock()
|
||||
aborted := u.aborted
|
||||
u.mu.Unlock()
|
||||
if aborted {
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
func (u *multipartUpload) setAudio(audio io.Closer) {
|
||||
u.mu.Lock()
|
||||
u.audio = audio
|
||||
aborted := u.aborted
|
||||
u.mu.Unlock()
|
||||
if aborted {
|
||||
_ = audio.Close()
|
||||
}
|
||||
}
|
||||
|
||||
func (u *multipartUpload) clearAudio(audio io.Closer) {
|
||||
u.mu.Lock()
|
||||
if u.audio == audio {
|
||||
u.audio = nil
|
||||
}
|
||||
u.mu.Unlock()
|
||||
}
|
||||
|
||||
func (u *multipartUpload) abort() {
|
||||
u.mu.Lock()
|
||||
if u.aborted {
|
||||
u.mu.Unlock()
|
||||
return
|
||||
}
|
||||
u.aborted = true
|
||||
audio := u.audio
|
||||
u.mu.Unlock()
|
||||
|
||||
_ = u.reader.Close()
|
||||
if audio != nil {
|
||||
_ = audio.Close()
|
||||
}
|
||||
}
|
||||
|
||||
type contextReader struct {
|
||||
ctx context.Context
|
||||
reader io.Reader
|
||||
}
|
||||
|
||||
func (r *contextReader) Read(p []byte) (int, error) {
|
||||
select {
|
||||
case <-r.ctx.Done():
|
||||
return 0, r.ctx.Err()
|
||||
default:
|
||||
return r.reader.Read(p)
|
||||
}
|
||||
}
|
||||
|
||||
func writeFileAtomic(path string, data []byte, perm os.FileMode) error {
|
||||
if strings.TrimSpace(path) == "" {
|
||||
return fmt.Errorf("path is required")
|
||||
}
|
||||
dir := filepath.Dir(path)
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
return fmt.Errorf("create parent dir %q: %w", dir, err)
|
||||
if err := fileops.WriteFileAtomic(path, data, perm); err != nil {
|
||||
return fmt.Errorf("write file %q: %w", path, err)
|
||||
}
|
||||
|
||||
base := filepath.Base(path)
|
||||
tmp, err := os.CreateTemp(dir, "."+base+".tmp-*")
|
||||
if err != nil {
|
||||
return fmt.Errorf("create temp file: %w", err)
|
||||
}
|
||||
tmpPath := tmp.Name()
|
||||
removeTmp := true
|
||||
defer func() {
|
||||
if removeTmp {
|
||||
_ = os.Remove(tmpPath)
|
||||
}
|
||||
}()
|
||||
|
||||
if _, err := tmp.Write(data); err != nil {
|
||||
_ = tmp.Close()
|
||||
return fmt.Errorf("write temp file: %w", err)
|
||||
}
|
||||
if err := tmp.Sync(); err != nil {
|
||||
_ = tmp.Close()
|
||||
return fmt.Errorf("sync temp file: %w", err)
|
||||
}
|
||||
if err := tmp.Close(); err != nil {
|
||||
return fmt.Errorf("close temp file: %w", err)
|
||||
}
|
||||
if err := os.Chmod(tmpPath, perm); err != nil {
|
||||
return fmt.Errorf("chmod temp file: %w", err)
|
||||
}
|
||||
if err := os.Rename(tmpPath, path); err != nil {
|
||||
return fmt.Errorf("rename temp file: %w", err)
|
||||
}
|
||||
removeTmp = false
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"io"
|
||||
"mime/multipart"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
@@ -19,6 +20,7 @@ func TestHTTPClientTranscribeSuccess(t *testing.T) {
|
||||
var gotLanguage string
|
||||
var gotFileField string
|
||||
var gotFileSize int
|
||||
var gotFileData string
|
||||
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method != http.MethodPost {
|
||||
@@ -40,6 +42,7 @@ func TestHTTPClientTranscribeSuccess(t *testing.T) {
|
||||
t.Fatalf("ReadAll(file) error = %v", err)
|
||||
}
|
||||
gotFileSize = len(data)
|
||||
gotFileData = string(data)
|
||||
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
_, _ = w.Write([]byte(`{"schema":"speaker_transcript.v1","segments":[]}`))
|
||||
@@ -77,13 +80,27 @@ func TestHTTPClientTranscribeSuccess(t *testing.T) {
|
||||
if gotFileSize == 0 {
|
||||
t.Fatal("file size = 0, want >0")
|
||||
}
|
||||
if gotFileData != "audio-data" {
|
||||
t.Fatalf("file data = %q, want exact payload", gotFileData)
|
||||
}
|
||||
verifyJSONFile(t, outPath)
|
||||
}
|
||||
|
||||
func TestHTTPClientRetriesOnTransientAndSucceeds(t *testing.T) {
|
||||
var calls atomic.Int32
|
||||
var payloads []string
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
n := calls.Add(1)
|
||||
file, _, err := r.FormFile("file")
|
||||
if err != nil {
|
||||
t.Fatalf("FormFile(file) error = %v", err)
|
||||
}
|
||||
data, err := io.ReadAll(file)
|
||||
_ = file.Close()
|
||||
if err != nil {
|
||||
t.Fatalf("ReadAll(file) error = %v", err)
|
||||
}
|
||||
payloads = append(payloads, string(data))
|
||||
if n == 1 {
|
||||
http.Error(w, "temporary", http.StatusInternalServerError)
|
||||
return
|
||||
@@ -112,6 +129,9 @@ func TestHTTPClientRetriesOnTransientAndSucceeds(t *testing.T) {
|
||||
if calls.Load() != 2 {
|
||||
t.Fatalf("calls = %d, want 2", calls.Load())
|
||||
}
|
||||
if len(payloads) != 2 || payloads[0] != "audio-data" || payloads[1] != "audio-data" {
|
||||
t.Fatalf("retry payloads = %#v, want two exact audio payloads", payloads)
|
||||
}
|
||||
verifyJSONFile(t, outPath)
|
||||
}
|
||||
|
||||
@@ -247,8 +267,245 @@ func TestHTTPClientConstructorValidation(t *testing.T) {
|
||||
if err == nil {
|
||||
t.Fatal("expected bad retry_delay error")
|
||||
}
|
||||
for _, endpoint := range []string{"ftp://example.com/transcribe", "file:///tmp/transcribe", "//example.com/transcribe", "https:/missing-host"} {
|
||||
if _, err := NewHTTPClientFromConfigValues(endpoint, "en", "30m", "2s", 1); err == nil {
|
||||
t.Errorf("NewHTTPClientFromConfigValues(%q) error = nil, want endpoint validation error", endpoint)
|
||||
}
|
||||
}
|
||||
for _, endpoint := range []string{"http://example.com/transcribe", "https://example.com/transcribe"} {
|
||||
if _, err := NewHTTPClientFromConfigValues(endpoint, "en", "30m", "2s", 1); err != nil {
|
||||
t.Errorf("NewHTTPClientFromConfigValues(%q) error = %v", endpoint, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestHTTPClientStreamsUploadBeforeSourceCompletes(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
source := newGatedReadCloser([]byte("audio-data"), release)
|
||||
firstByteReceived := make(chan struct{})
|
||||
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
part := firstMultipartFilePart(t, r)
|
||||
buf := make([]byte, 1)
|
||||
if _, err := part.Read(buf); err != nil {
|
||||
t.Errorf("Read(file) error = %v", err)
|
||||
return
|
||||
}
|
||||
close(firstByteReceived)
|
||||
if _, err := io.Copy(io.Discard, part); err != nil {
|
||||
t.Errorf("discard remaining file data: %v", err)
|
||||
return
|
||||
}
|
||||
_, _ = w.Write([]byte(`{"ok":true}`))
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
client := newTestHTTPClient(t, srv.URL)
|
||||
client.openAudio = func(string) (io.ReadCloser, error) { return source, nil }
|
||||
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
_, err := client.Transcribe(context.Background(), TranscribeRequest{AudioPath: "audio.flac", OutputRawTranscriptPath: filepath.Join(t.TempDir(), "raw.json")})
|
||||
done <- err
|
||||
}()
|
||||
|
||||
select {
|
||||
case <-firstByteReceived:
|
||||
close(release)
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("server did not receive streamed audio before source completed")
|
||||
}
|
||||
if err := <-done; err != nil {
|
||||
t.Fatalf("Transcribe() error = %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHTTPClientSourceReadFailureReachesCaller(t *testing.T) {
|
||||
sourceErr := errors.New("source read failed")
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
_, _ = io.Copy(io.Discard, r.Body)
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
client := newTestHTTPClient(t, srv.URL)
|
||||
client.openAudio = func(string) (io.ReadCloser, error) {
|
||||
return &failingReadCloser{first: []byte("partial"), err: sourceErr}, nil
|
||||
}
|
||||
|
||||
_, err := client.Transcribe(context.Background(), TranscribeRequest{AudioPath: "audio.flac", OutputRawTranscriptPath: filepath.Join(t.TempDir(), "raw.json")})
|
||||
if !errors.Is(err, sourceErr) {
|
||||
t.Fatalf("Transcribe() error = %v, want source read failure", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHTTPClientEarlyServerResponseReturns(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
source := newGatedReadCloser([]byte("audio-data"), release)
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
http.Error(w, "bad request", http.StatusBadRequest)
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
client := newTestHTTPClient(t, srv.URL)
|
||||
client.openAudio = func(string) (io.ReadCloser, error) { return source, nil }
|
||||
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
_, err := client.Transcribe(context.Background(), TranscribeRequest{AudioPath: "audio.flac", OutputRawTranscriptPath: filepath.Join(t.TempDir(), "raw.json")})
|
||||
done <- err
|
||||
}()
|
||||
|
||||
select {
|
||||
case err := <-done:
|
||||
if err == nil {
|
||||
t.Fatal("Transcribe() error = nil, want HTTP status error")
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("Transcribe() did not finish after server closed the request early")
|
||||
}
|
||||
}
|
||||
|
||||
func TestHTTPClientCancellationReleasesBlockedProducer(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
source := newGatedReadCloser([]byte("audio-data"), release)
|
||||
firstByteReceived := make(chan struct{})
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
part := firstMultipartFilePart(t, r)
|
||||
buf := make([]byte, 1)
|
||||
if _, err := part.Read(buf); err != nil {
|
||||
t.Errorf("Read(file) error = %v", err)
|
||||
return
|
||||
}
|
||||
close(firstByteReceived)
|
||||
select {
|
||||
case <-r.Context().Done():
|
||||
case <-source.closed:
|
||||
}
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
client := newTestHTTPClient(t, srv.URL)
|
||||
client.openAudio = func(string) (io.ReadCloser, error) { return source, nil }
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
done := make(chan error, 1)
|
||||
go func() {
|
||||
_, err := client.Transcribe(ctx, TranscribeRequest{AudioPath: "audio.flac", OutputRawTranscriptPath: filepath.Join(t.TempDir(), "raw.json")})
|
||||
done <- err
|
||||
}()
|
||||
|
||||
select {
|
||||
case <-firstByteReceived:
|
||||
cancel()
|
||||
case <-time.After(time.Second):
|
||||
cancel()
|
||||
t.Fatal("server did not receive initial streamed audio")
|
||||
}
|
||||
select {
|
||||
case err := <-done:
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("Transcribe() error = %v, want context cancellation", err)
|
||||
}
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("Transcribe() did not finish after cancellation")
|
||||
}
|
||||
select {
|
||||
case <-source.closed:
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("blocked audio source was not closed on cancellation")
|
||||
}
|
||||
}
|
||||
|
||||
func TestHTTPClientBoundsWhisperXResponse(t *testing.T) {
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
_, _ = io.Copy(io.Discard, r.Body)
|
||||
_, _ = w.Write([]byte(`{"ok":true}`))
|
||||
}))
|
||||
defer srv.Close()
|
||||
|
||||
client, err := NewHTTPClient(HTTPClientConfig{TranscribeURL: srv.URL, Language: "en", Timeout: time.Second, MaxResponseBytes: 4})
|
||||
if err != nil {
|
||||
t.Fatalf("NewHTTPClient() error = %v", err)
|
||||
}
|
||||
audioPath := writeWhisperXTestFile(t, "audio.flac", "audio-data")
|
||||
_, err = client.Transcribe(context.Background(), TranscribeRequest{AudioPath: audioPath, OutputRawTranscriptPath: filepath.Join(t.TempDir(), "raw.json")})
|
||||
if err == nil || !strings.Contains(err.Error(), "whisperx response exceeds configured limit") {
|
||||
t.Fatalf("Transcribe() error = %v, want bounded WhisperX response error", err)
|
||||
}
|
||||
}
|
||||
|
||||
func newTestHTTPClient(t *testing.T, endpoint string) *HTTPClient {
|
||||
t.Helper()
|
||||
client, err := NewHTTPClientFromConfigValues(endpoint, "en", "2s", "1ms", 0)
|
||||
if err != nil {
|
||||
t.Fatalf("NewHTTPClientFromConfigValues() error = %v", err)
|
||||
}
|
||||
return client
|
||||
}
|
||||
|
||||
func firstMultipartFilePart(t *testing.T, r *http.Request) *multipart.Part {
|
||||
t.Helper()
|
||||
reader, err := r.MultipartReader()
|
||||
if err != nil {
|
||||
t.Fatalf("MultipartReader() error = %v", err)
|
||||
}
|
||||
part, err := reader.NextPart()
|
||||
if err != nil {
|
||||
t.Fatalf("NextPart() error = %v", err)
|
||||
}
|
||||
if part.FormName() != "file" {
|
||||
t.Fatalf("first form field = %q, want file", part.FormName())
|
||||
}
|
||||
return part
|
||||
}
|
||||
|
||||
type gatedReadCloser struct {
|
||||
first []byte
|
||||
release <-chan struct{}
|
||||
closed chan struct{}
|
||||
sent bool
|
||||
once atomic.Bool
|
||||
}
|
||||
|
||||
func newGatedReadCloser(first []byte, release <-chan struct{}) *gatedReadCloser {
|
||||
return &gatedReadCloser{first: first, release: release, closed: make(chan struct{})}
|
||||
}
|
||||
|
||||
func (r *gatedReadCloser) Read(p []byte) (int, error) {
|
||||
if !r.sent {
|
||||
r.sent = true
|
||||
return copy(p, r.first), nil
|
||||
}
|
||||
select {
|
||||
case <-r.release:
|
||||
return 0, io.EOF
|
||||
case <-r.closed:
|
||||
return 0, errors.New("audio source closed")
|
||||
}
|
||||
}
|
||||
|
||||
func (r *gatedReadCloser) Close() error {
|
||||
if r.once.CompareAndSwap(false, true) {
|
||||
close(r.closed)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
type failingReadCloser struct {
|
||||
first []byte
|
||||
err error
|
||||
sent bool
|
||||
}
|
||||
|
||||
func (r *failingReadCloser) Read(p []byte) (int, error) {
|
||||
if !r.sent {
|
||||
r.sent = true
|
||||
return copy(p, r.first), nil
|
||||
}
|
||||
return 0, r.err
|
||||
}
|
||||
|
||||
func (r *failingReadCloser) Close() error { return nil }
|
||||
|
||||
func writeWhisperXTestFile(t *testing.T, name, contents string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(t.TempDir(), name)
|
||||
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
)
|
||||
|
||||
@@ -47,19 +48,55 @@ func (f *artifactSelectionFlag) Normalize() ([]string, error) {
|
||||
}
|
||||
|
||||
func validateSelectedArtifacts(cfg *config.Config, selected []string) error {
|
||||
if len(selected) == 0 {
|
||||
return nil
|
||||
}
|
||||
_, err := resolveEffectiveArtifacts(cfg, selected)
|
||||
return err
|
||||
}
|
||||
|
||||
func resolveEffectiveArtifacts(cfg *config.Config, selected []string) (artifacts.EffectiveArtifactSet, error) {
|
||||
if cfg == nil || cfg.Pipeline == nil || cfg.Pipeline.Scriptorium == nil {
|
||||
return fmt.Errorf("--artifacts requires pipeline.scriptorium.artifacts to be configured")
|
||||
if len(selected) == 0 {
|
||||
return artifacts.ResolveEffectiveArtifactSet(nil, nil)
|
||||
}
|
||||
return artifacts.EffectiveArtifactSet{}, fmt.Errorf("--artifacts requires pipeline.scriptorium.artifacts to be configured")
|
||||
}
|
||||
configured := cfg.Pipeline.Scriptorium.Artifacts
|
||||
if len(configured) == 0 {
|
||||
return fmt.Errorf("--artifacts requires at least one configured artifact in pipeline.scriptorium.artifacts")
|
||||
configured := artifacts.ConfiguredArtifactDefinitions(cfg.Pipeline.Scriptorium.Artifacts)
|
||||
if len(selected) > 0 && len(configured) == 0 {
|
||||
return artifacts.EffectiveArtifactSet{}, fmt.Errorf("--artifacts requires at least one configured artifact in pipeline.scriptorium.artifacts")
|
||||
}
|
||||
for _, name := range selected {
|
||||
if _, ok := configured[name]; !ok {
|
||||
return fmt.Errorf("--artifacts includes unknown artifact %q", name)
|
||||
effective, err := artifacts.ResolveEffectiveArtifactSet(configured, selected)
|
||||
if err != nil {
|
||||
if strings.Contains(err.Error(), "is not configured") {
|
||||
return artifacts.EffectiveArtifactSet{}, fmt.Errorf("--artifacts includes unknown artifact %q", selectedArtifactName(err))
|
||||
}
|
||||
return artifacts.EffectiveArtifactSet{}, err
|
||||
}
|
||||
if err := validateEffectiveArtifactConfiguration(cfg.Pipeline.Scriptorium.Artifacts, effective); err != nil {
|
||||
return artifacts.EffectiveArtifactSet{}, err
|
||||
}
|
||||
return effective, nil
|
||||
}
|
||||
|
||||
func selectedArtifactName(err error) string {
|
||||
message := err.Error()
|
||||
start := strings.Index(message, "\"")
|
||||
if start < 0 {
|
||||
return ""
|
||||
}
|
||||
end := strings.Index(message[start+1:], "\"")
|
||||
if end < 0 {
|
||||
return ""
|
||||
}
|
||||
return message[start+1 : start+1+end]
|
||||
}
|
||||
|
||||
func validateEffectiveArtifactConfiguration(configured map[string]config.ScriptoriumArtifactConfig, effective artifacts.EffectiveArtifactSet) error {
|
||||
for _, name := range effective.Keys() {
|
||||
artifactCfg := configured[name]
|
||||
if strings.TrimSpace(artifactCfg.PromptID) == "" {
|
||||
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.prompt_id is required when selected", name)
|
||||
}
|
||||
if strings.TrimSpace(artifactCfg.OutputPath) == "" {
|
||||
return fmt.Errorf("pipeline.scriptorium.artifacts.%s.output_path is required when selected", name)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
|
||||
@@ -11,7 +11,6 @@ import (
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/stage"
|
||||
)
|
||||
|
||||
func TestExecuteRunStageArtifactsUnsupportedStageFails(t *testing.T) {
|
||||
@@ -43,8 +42,8 @@ func TestExecuteRunStagePublishPropagatesSelectedArtifacts(t *testing.T) {
|
||||
t.Cleanup(func() {
|
||||
executeStagesFn = origExecuteStagesFn
|
||||
})
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
|
||||
for _, s := range stages {
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, plan BoundedPlan, opts RunOptions) (*RunSummary, error) {
|
||||
for _, s := range plan.Stages() {
|
||||
capturedStages = append(capturedStages, s.Name())
|
||||
}
|
||||
capturedArtifacts = append([]string(nil), opts.SelectedArtifacts...)
|
||||
@@ -115,8 +114,8 @@ func TestRunStageArtifactsDoesNotImplyForce(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("RunStage() error = %v", err)
|
||||
}
|
||||
if !strings.Contains(out.String(), "stage=analyze executed=0 skipped=1 force=false") {
|
||||
t.Fatalf("output = %q, want analyze skip without force", out.String())
|
||||
if !strings.Contains(out.String(), "stage=analyze executed=1 skipped=0 force=false") {
|
||||
t.Fatalf("output = %q, want legacy analyze evidence rebuilt without implying force", out.String())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -144,8 +143,8 @@ func TestRunArtifactsWithSucceededAnalyzeSkipsUnlessForced(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("Run() error = %v", err)
|
||||
}
|
||||
if !strings.Contains(out.String(), "executed=1 skipped=11") {
|
||||
t.Fatalf("output = %q, want all stages skipped", out.String())
|
||||
if !strings.Contains(out.String(), "executed=4 skipped=8") {
|
||||
t.Fatalf("output = %q, want extract reconsidered and legacy analyze plus delivery rebuilt", out.String())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -159,8 +158,8 @@ func TestExecuteAnalyzeForceRunsAnalyze(t *testing.T) {
|
||||
t.Cleanup(func() {
|
||||
executeStagesFn = origExecuteStagesFn
|
||||
})
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
|
||||
for _, s := range stages {
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, plan BoundedPlan, opts RunOptions) (*RunSummary, error) {
|
||||
for _, s := range plan.Stages() {
|
||||
capturedStages = append(capturedStages, s.Name())
|
||||
}
|
||||
capturedForce = opts.Force
|
||||
@@ -200,7 +199,7 @@ func TestExecuteAnalyzePropagatesSelectedArtifacts(t *testing.T) {
|
||||
t.Cleanup(func() {
|
||||
executeStagesFn = origExecuteStagesFn
|
||||
})
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, _ []stage.Stage, opts RunOptions) (*RunSummary, error) {
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, _ BoundedPlan, opts RunOptions) (*RunSummary, error) {
|
||||
capturedArtifacts = append([]string(nil), opts.SelectedArtifacts...)
|
||||
return &RunSummary{ManifestPath: filepath.Join(workspaceRoot, "manifest.json"), Executed: []string{"analyze"}}, nil
|
||||
}
|
||||
@@ -293,8 +292,8 @@ func TestExecutePublishForceRunsPublish(t *testing.T) {
|
||||
t.Cleanup(func() {
|
||||
executeStagesFn = origExecuteStagesFn
|
||||
})
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, stages []stage.Stage, opts RunOptions) (*RunSummary, error) {
|
||||
for _, s := range stages {
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, plan BoundedPlan, opts RunOptions) (*RunSummary, error) {
|
||||
for _, s := range plan.Stages() {
|
||||
capturedStages = append(capturedStages, s.Name())
|
||||
}
|
||||
capturedForce = opts.Force
|
||||
|
||||
@@ -110,6 +110,20 @@ func TestValidateSelectedArtifacts(t *testing.T) {
|
||||
},
|
||||
selected: []string{"player_handout", "session_recap"},
|
||||
},
|
||||
{
|
||||
name: "selected disabled artifact must be executable",
|
||||
cfg: &config.Config{
|
||||
Pipeline: &config.PipelineConfig{
|
||||
Scriptorium: &config.ScriptoriumConfig{
|
||||
Artifacts: map[string]config.ScriptoriumArtifactConfig{
|
||||
"player_handout": {Enabled: false, OutputPath: "artifacts/player_handout.md"},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
selected: []string{"player_handout"},
|
||||
wantErr: "pipeline.scriptorium.artifacts.player_handout.prompt_id is required when selected",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
|
||||
37
internal/app/analyze_evidence_test_helpers_test.go
Normal file
37
internal/app/analyze_evidence_test_helpers_test.go
Normal file
@@ -0,0 +1,37 @@
|
||||
package app
|
||||
|
||||
import (
|
||||
"crypto/sha256"
|
||||
"encoding/hex"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifactmodel"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||
)
|
||||
|
||||
func setAppAnalyzeEvidence(m *manifest.Manifest, key, relativePath string, body []byte) {
|
||||
now := time.Date(2026, 5, 19, 23, 0, 0, 0, time.UTC)
|
||||
record := m.Stages["analyze"]
|
||||
if record == nil {
|
||||
record = &manifest.StageRecord{Name: "analyze", Status: manifest.StatusSucceeded, CreatedAt: now, UpdatedAt: now}
|
||||
m.Stages["analyze"] = record
|
||||
}
|
||||
if record.AnalyzeArtifacts == nil {
|
||||
record.AnalyzeArtifacts = map[string]manifest.AnalyzeArtifactRecord{}
|
||||
}
|
||||
digest := sha256.Sum256(body)
|
||||
record.AnalyzeStateVersion = manifest.AnalyzeStateContractVersion
|
||||
record.AnalyzeArtifacts[key] = manifest.AnalyzeArtifactRecord{
|
||||
Key: key, Status: manifest.AnalyzeArtifactCurrent,
|
||||
FingerprintVersion: manifest.AnalyzeFingerprintContractVersion,
|
||||
Fingerprint: strings.Repeat("1", 64),
|
||||
Output: &manifest.ArtifactRecord{
|
||||
Kind: "scriptorium_artifact", SourceID: artifacts.ConfiguredArtifactSourceID(key), LocalPath: relativePath,
|
||||
Contract: &artifactmodel.ContractMetadata{MediaType: "text/markdown", SchemaID: "narratio." + key, SchemaVersion: "1"},
|
||||
ProducerRunID: "run-1", Checksum: hex.EncodeToString(digest[:]),
|
||||
},
|
||||
OutputSize: int64(len(body)), ProducerRunID: "run-1", UpdatedAt: now,
|
||||
}
|
||||
}
|
||||
134
internal/app/analyze_projection.go
Normal file
134
internal/app/analyze_projection.go
Normal file
@@ -0,0 +1,134 @@
|
||||
package app
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"reflect"
|
||||
"sort"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/stage"
|
||||
)
|
||||
|
||||
type validatedAnalyzeProjection struct {
|
||||
session map[string]manifest.AnalyzeArtifactRecord
|
||||
invocation map[string]manifest.AnalyzeArtifactRecord
|
||||
}
|
||||
|
||||
type analyzeStateSnapshot struct {
|
||||
version int
|
||||
records map[string]manifest.AnalyzeArtifactRecord
|
||||
}
|
||||
|
||||
func captureAnalyzeState(manifestValue *manifest.Manifest, stageName string) analyzeStateSnapshot {
|
||||
if manifestValue == nil || stageName != "analyze" || manifestValue.Stages["analyze"] == nil {
|
||||
return analyzeStateSnapshot{}
|
||||
}
|
||||
record := manifestValue.Stages["analyze"]
|
||||
return analyzeStateSnapshot{
|
||||
version: record.AnalyzeStateVersion,
|
||||
records: manifest.CloneAnalyzeArtifactCollection(record.AnalyzeArtifacts),
|
||||
}
|
||||
}
|
||||
|
||||
func restoreAnalyzeState(manifestValue *manifest.Manifest, snapshot analyzeStateSnapshot) {
|
||||
if manifestValue == nil || manifestValue.Stages["analyze"] == nil {
|
||||
return
|
||||
}
|
||||
record := manifestValue.Stages["analyze"]
|
||||
record.AnalyzeStateVersion = snapshot.version
|
||||
record.AnalyzeArtifacts = manifest.CloneAnalyzeArtifactCollection(snapshot.records)
|
||||
}
|
||||
|
||||
func validateSuccessfulAnalyzeProjection(stageName string, result *stage.StageResult) (*validatedAnalyzeProjection, error) {
|
||||
if result == nil || result.AnalyzeState == nil {
|
||||
return nil, nil
|
||||
}
|
||||
if stageName != "analyze" {
|
||||
return nil, fmt.Errorf("stage %q returned analyze-owned state projection", stageName)
|
||||
}
|
||||
if result.Disposition == stage.StageDispositionSkipped {
|
||||
return nil, fmt.Errorf("skipped analyze result cannot contain analyze-owned state projection")
|
||||
}
|
||||
if len(result.Outputs) != 0 {
|
||||
return nil, fmt.Errorf("analyze result with state projection cannot contain ordinary outputs")
|
||||
}
|
||||
return validateAndCloneAnalyzeProjection(result.AnalyzeState)
|
||||
}
|
||||
|
||||
func validateFailedAnalyzeProjection(stageName string, result *stage.StageResult) (*validatedAnalyzeProjection, error) {
|
||||
if result == nil || result.AnalyzeState == nil {
|
||||
return nil, nil
|
||||
}
|
||||
if stageName != "analyze" {
|
||||
return nil, fmt.Errorf("stage %q returned analyze-owned state projection with an error", stageName)
|
||||
}
|
||||
if result.Disposition != stage.StageDispositionSucceeded || result.SkipReason != "" || len(result.Outputs) != 0 || len(result.Logs) != 0 || len(result.GeneratedConfigs) != 0 || len(result.Metadata) != 0 {
|
||||
return nil, fmt.Errorf("analyze result with an error may contain only analyze-owned state projection")
|
||||
}
|
||||
return validateAndCloneAnalyzeProjection(result.AnalyzeState)
|
||||
}
|
||||
|
||||
func validateAndCloneAnalyzeProjection(projection *stage.AnalyzeStateProjection) (*validatedAnalyzeProjection, error) {
|
||||
session := manifest.CloneAnalyzeArtifactCollection(projection.Session)
|
||||
invocation := manifest.CloneAnalyzeArtifactCollection(projection.Invocation)
|
||||
if err := manifest.ValidateAnalyzeArtifactCollection(manifest.AnalyzeStateContractVersion, session); err != nil {
|
||||
return nil, fmt.Errorf("validate reconciled session analyze state: %w", err)
|
||||
}
|
||||
if err := manifest.ValidateAnalyzeArtifactCollection(manifest.AnalyzeStateContractVersion, invocation); err != nil {
|
||||
return nil, fmt.Errorf("validate invocation analyze state: %w", err)
|
||||
}
|
||||
for key, invocationRecord := range invocation {
|
||||
sessionRecord, ok := session[key]
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("invocation analyze artifact %q is absent from reconciled session state", key)
|
||||
}
|
||||
if !reflect.DeepEqual(invocationRecord, sessionRecord) {
|
||||
return nil, fmt.Errorf("invocation analyze artifact %q contradicts reconciled session state", key)
|
||||
}
|
||||
}
|
||||
return &validatedAnalyzeProjection{session: session, invocation: invocation}, nil
|
||||
}
|
||||
|
||||
func applyAnalyzeProjection(
|
||||
sessionManifest *manifest.Manifest,
|
||||
runManifest *manifest.RunManifest,
|
||||
projection *validatedAnalyzeProjection,
|
||||
) {
|
||||
if projection == nil {
|
||||
return
|
||||
}
|
||||
if sessionManifest != nil && sessionManifest.Stages["analyze"] != nil {
|
||||
record := sessionManifest.Stages["analyze"]
|
||||
record.AnalyzeStateVersion = manifest.AnalyzeStateContractVersion
|
||||
record.AnalyzeArtifacts = manifest.CloneAnalyzeArtifactCollection(projection.session)
|
||||
}
|
||||
if runManifest != nil && runManifest.Stages["analyze"] != nil {
|
||||
record := runManifest.Stages["analyze"]
|
||||
record.AnalyzeStateVersion = manifest.AnalyzeStateContractVersion
|
||||
record.AnalyzeArtifacts = manifest.CloneAnalyzeArtifactCollection(projection.invocation)
|
||||
}
|
||||
}
|
||||
|
||||
func analyzeProjectionOutputs(records map[string]manifest.AnalyzeArtifactRecord, producerRunID string) []manifest.ArtifactRecord {
|
||||
keys := make([]string, 0, len(records))
|
||||
for key, record := range records {
|
||||
if record.Status != manifest.AnalyzeArtifactCurrent || record.Output == nil {
|
||||
continue
|
||||
}
|
||||
if producerRunID != "" && record.ProducerRunID != producerRunID {
|
||||
continue
|
||||
}
|
||||
keys = append(keys, key)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
outputs := make([]manifest.ArtifactRecord, 0, len(keys))
|
||||
for _, key := range keys {
|
||||
record := manifest.CloneAnalyzeArtifactCollection(map[string]manifest.AnalyzeArtifactRecord{key: records[key]})[key]
|
||||
output := *record.Output
|
||||
if output.ProducerRunID == "" {
|
||||
output.ProducerRunID = record.ProducerRunID
|
||||
}
|
||||
outputs = append(outputs, output)
|
||||
}
|
||||
return outputs
|
||||
}
|
||||
346
internal/app/analyze_projection_test.go
Normal file
346
internal/app/analyze_projection_test.go
Normal file
@@ -0,0 +1,346 @@
|
||||
package app
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifactmodel"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/stage"
|
||||
)
|
||||
|
||||
type projectionStage struct {
|
||||
name string
|
||||
run func(*stage.Env, *manifest.Manifest) (*stage.StageResult, error)
|
||||
}
|
||||
|
||||
func (s projectionStage) Name() string { return s.name }
|
||||
func (s projectionStage) Run(_ context.Context, env *stage.Env, m *manifest.Manifest) (*stage.StageResult, error) {
|
||||
return s.run(env, m)
|
||||
}
|
||||
|
||||
func TestExecuteStagesProjectsSeparateSessionAndInvocationAnalyzeState(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
oldAt := time.Date(2026, 5, 3, 9, 0, 0, 0, time.UTC)
|
||||
oldRecord := appAnalyzeRecord("player_handout", manifest.AnalyzeArtifactCurrent, "old-run", oldAt)
|
||||
|
||||
stageToRun := projectionStage{name: "analyze", run: func(_ *stage.Env, m *manifest.Manifest) (*stage.StageResult, error) {
|
||||
newRecord := appAnalyzeRecord("session_recap", manifest.AnalyzeArtifactCurrent, m.RunID, time.Now().UTC())
|
||||
staleRecord := appAnalyzeRecord("quest_log", manifest.AnalyzeArtifactStale, m.RunID, time.Now().UTC())
|
||||
session := map[string]manifest.AnalyzeArtifactRecord{
|
||||
"player_handout": oldRecord,
|
||||
"quest_log": staleRecord,
|
||||
"session_recap": newRecord,
|
||||
}
|
||||
return &stage.StageResult{
|
||||
Logs: []string{"aggregate-analyze.log"},
|
||||
AnalyzeState: &stage.AnalyzeStateProjection{
|
||||
Session: session,
|
||||
Invocation: map[string]manifest.AnalyzeArtifactRecord{
|
||||
"player_handout": oldRecord,
|
||||
"session_recap": newRecord,
|
||||
},
|
||||
},
|
||||
}, nil
|
||||
}}
|
||||
|
||||
summary, err := executeStages(context.Background(), cfg, []stage.Stage{stageToRun}, RunOptions{})
|
||||
if err != nil {
|
||||
t.Fatalf("executeStages() error = %v", err)
|
||||
}
|
||||
store := &manifest.LocalStore{}
|
||||
sessionManifest, err := store.Load(context.Background(), summary.ManifestPath)
|
||||
if err != nil {
|
||||
t.Fatalf("Load(session) error = %v", err)
|
||||
}
|
||||
analyze := sessionManifest.Stages["analyze"]
|
||||
if analyze.AnalyzeStateVersion != manifest.AnalyzeStateContractVersion || len(analyze.AnalyzeArtifacts) != 3 {
|
||||
t.Fatalf("session analyze state = %#v", analyze)
|
||||
}
|
||||
if got := analyzeArtifactOutputKeys(analyze.Outputs); !reflect.DeepEqual(got, []string{"player_handout", "session_recap"}) {
|
||||
t.Fatalf("session aggregate outputs = %#v, want current records only", got)
|
||||
}
|
||||
if len(analyze.Logs) != 1 || analyze.Logs[0] != "aggregate-analyze.log" {
|
||||
t.Fatalf("session aggregate logs = %#v", analyze.Logs)
|
||||
}
|
||||
|
||||
runManifest, err := store.LoadRun(context.Background(), summary.RunManifestPath)
|
||||
if err != nil {
|
||||
t.Fatalf("LoadRun() error = %v", err)
|
||||
}
|
||||
runAnalyze := runManifest.Stages["analyze"]
|
||||
if len(runAnalyze.AnalyzeArtifacts) != 2 || runAnalyze.AnalyzeArtifacts["session_recap"].Status != manifest.AnalyzeArtifactCurrent {
|
||||
t.Fatalf("invocation analyze state = %#v", runAnalyze.AnalyzeArtifacts)
|
||||
}
|
||||
if got := analyzeArtifactOutputKeys(runAnalyze.Outputs); !reflect.DeepEqual(got, []string{"session_recap"}) {
|
||||
t.Fatalf("invocation outputs = %#v, want produced artifact only", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteStagesPersistsRestrictedAnalyzeStateOnPartialError(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
now := time.Date(2026, 5, 3, 9, 0, 0, 0, time.UTC)
|
||||
seed := manifest.New(cfg.Session.SessionID, now)
|
||||
seed.Campaign = cfg.Session.Campaign
|
||||
seed.MarkStageSucceeded("analyze", now, nil)
|
||||
seed.Stages["analyze"].AnalyzeStateVersion = manifest.AnalyzeStateContractVersion
|
||||
seed.Stages["analyze"].AnalyzeArtifacts = map[string]manifest.AnalyzeArtifactRecord{
|
||||
"player_handout": appAnalyzeRecord("player_handout", manifest.AnalyzeArtifactCurrent, "old-run", now),
|
||||
}
|
||||
seed.MarkStageSucceeded("publish", now, nil)
|
||||
saveBoundedManifest(t, cfg, seed)
|
||||
|
||||
stageToRun := projectionStage{name: "analyze", run: func(_ *stage.Env, m *manifest.Manifest) (*stage.StageResult, error) {
|
||||
unrelated := seed.Stages["analyze"].AnalyzeArtifacts["player_handout"]
|
||||
completed := appAnalyzeRecord("session_recap", manifest.AnalyzeArtifactCurrent, m.RunID, time.Now().UTC())
|
||||
failed := appAnalyzeRecord("quest_log", manifest.AnalyzeArtifactFailed, m.RunID, time.Now().UTC())
|
||||
session := map[string]manifest.AnalyzeArtifactRecord{
|
||||
"player_handout": unrelated,
|
||||
"quest_log": failed,
|
||||
"session_recap": completed,
|
||||
}
|
||||
return &stage.StageResult{AnalyzeState: &stage.AnalyzeStateProjection{
|
||||
Session: session,
|
||||
Invocation: map[string]manifest.AnalyzeArtifactRecord{
|
||||
"quest_log": failed,
|
||||
"session_recap": completed,
|
||||
},
|
||||
}}, errors.New("quest log failed")
|
||||
}}
|
||||
|
||||
_, err := executeStages(context.Background(), cfg, []stage.Stage{stageToRun}, RunOptions{Force: true})
|
||||
if err == nil || !strings.Contains(err.Error(), "quest log failed") {
|
||||
t.Fatalf("executeStages() error = %v", err)
|
||||
}
|
||||
store := &manifest.LocalStore{}
|
||||
loaded, err := store.Load(context.Background(), manifestPathFor(cfg))
|
||||
if err != nil {
|
||||
t.Fatalf("Load(session) error = %v", err)
|
||||
}
|
||||
analyze := loaded.Stages["analyze"]
|
||||
if analyze.Status != manifest.StatusFailed || len(analyze.Outputs) != 0 {
|
||||
t.Fatalf("aggregate analyze state = %#v, want failed without outputs", analyze)
|
||||
}
|
||||
if analyze.AnalyzeArtifacts["player_handout"].Status != manifest.AnalyzeArtifactCurrent ||
|
||||
analyze.AnalyzeArtifacts["session_recap"].Status != manifest.AnalyzeArtifactCurrent ||
|
||||
analyze.AnalyzeArtifacts["quest_log"].Status != manifest.AnalyzeArtifactFailed {
|
||||
t.Fatalf("partial session projection = %#v", analyze.AnalyzeArtifacts)
|
||||
}
|
||||
if loaded.Stages["publish"].Status != manifest.StatusStale {
|
||||
t.Fatalf("publish status = %q, want stale", loaded.Stages["publish"].Status)
|
||||
}
|
||||
|
||||
runsDir := artifacts.SessionRunsDirForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID)
|
||||
entries, err := os.ReadDir(runsDir)
|
||||
if err != nil || len(entries) != 1 {
|
||||
t.Fatalf("run directory entries = %#v, error = %v", entries, err)
|
||||
}
|
||||
runManifest, err := store.LoadRun(context.Background(), filepath.Join(runsDir, entries[0].Name(), "manifest.json"))
|
||||
if err != nil {
|
||||
t.Fatalf("LoadRun() error = %v", err)
|
||||
}
|
||||
runAnalyze := runManifest.Stages["analyze"]
|
||||
if runAnalyze.Status != manifest.StatusFailed || len(runAnalyze.AnalyzeArtifacts) != 2 || len(runAnalyze.Outputs) != 0 {
|
||||
t.Fatalf("partial invocation projection = %#v", runAnalyze)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteStagesRejectsInvalidAnalyzeProjectionWithoutReplacingPriorState(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
now := time.Date(2026, 5, 3, 9, 0, 0, 0, time.UTC)
|
||||
prior := appAnalyzeRecord("player_handout", manifest.AnalyzeArtifactCurrent, "old-run", now)
|
||||
seed := manifest.New(cfg.Session.SessionID, now)
|
||||
seed.Campaign = cfg.Session.Campaign
|
||||
seed.MarkStageSucceeded("analyze", now, nil)
|
||||
seed.Stages["analyze"].AnalyzeStateVersion = manifest.AnalyzeStateContractVersion
|
||||
seed.Stages["analyze"].AnalyzeArtifacts = map[string]manifest.AnalyzeArtifactRecord{"player_handout": prior}
|
||||
saveBoundedManifest(t, cfg, seed)
|
||||
|
||||
stageToRun := projectionStage{name: "analyze", run: func(_ *stage.Env, m *manifest.Manifest) (*stage.StageResult, error) {
|
||||
invalid := appAnalyzeRecord("session_recap", manifest.AnalyzeArtifactCurrent, m.RunID, time.Now().UTC())
|
||||
invalid.Output.Checksum = "invalid"
|
||||
return &stage.StageResult{AnalyzeState: &stage.AnalyzeStateProjection{
|
||||
Session: map[string]manifest.AnalyzeArtifactRecord{"session_recap": invalid},
|
||||
Invocation: map[string]manifest.AnalyzeArtifactRecord{"session_recap": invalid},
|
||||
}}, errors.New("analysis failed")
|
||||
}}
|
||||
|
||||
_, err := executeStages(context.Background(), cfg, []stage.Stage{stageToRun}, RunOptions{Force: true})
|
||||
if err == nil || !strings.Contains(err.Error(), "checksum") {
|
||||
t.Fatalf("executeStages() error = %v, want projection validation failure", err)
|
||||
}
|
||||
loaded, loadErr := (&manifest.LocalStore{}).Load(context.Background(), manifestPathFor(cfg))
|
||||
if loadErr != nil {
|
||||
t.Fatalf("Load() error = %v", loadErr)
|
||||
}
|
||||
if len(loaded.Stages["analyze"].AnalyzeArtifacts) != 1 || !reflect.DeepEqual(loaded.Stages["analyze"].AnalyzeArtifacts["player_handout"], prior) {
|
||||
t.Fatalf("prior state replaced by invalid projection: %#v", loaded.Stages["analyze"].AnalyzeArtifacts)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteStagesRollsBackAnalyzeAuthorityWhenProjectionSaveFails(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
now := time.Date(2026, 5, 3, 9, 0, 0, 0, time.UTC)
|
||||
prior := appAnalyzeRecord("player_handout", manifest.AnalyzeArtifactCurrent, "old-run", now)
|
||||
seed := manifest.New(cfg.Session.SessionID, now)
|
||||
seed.Campaign = cfg.Session.Campaign
|
||||
seed.MarkStageSucceeded("analyze", now, nil)
|
||||
seed.Stages["analyze"].AnalyzeStateVersion = manifest.AnalyzeStateContractVersion
|
||||
seed.Stages["analyze"].AnalyzeArtifacts = map[string]manifest.AnalyzeArtifactRecord{"player_handout": prior}
|
||||
saveBoundedManifest(t, cfg, seed)
|
||||
|
||||
stageReturned := false
|
||||
store := &analyzeProjectionFailingStore{delegate: &manifest.LocalStore{}, shouldFail: func(m *manifest.Manifest) bool {
|
||||
return stageReturned && m.Stages["analyze"] != nil && m.Stages["analyze"].Status == manifest.StatusSucceeded && m.Stages["analyze"].AnalyzeArtifacts["session_recap"].Status == manifest.AnalyzeArtifactCurrent
|
||||
}}
|
||||
stageToRun := projectionStage{name: "analyze", run: func(_ *stage.Env, m *manifest.Manifest) (*stage.StageResult, error) {
|
||||
stageReturned = true
|
||||
newRecord := appAnalyzeRecord("session_recap", manifest.AnalyzeArtifactCurrent, m.RunID, time.Now().UTC())
|
||||
return &stage.StageResult{AnalyzeState: &stage.AnalyzeStateProjection{
|
||||
Session: map[string]manifest.AnalyzeArtifactRecord{
|
||||
"player_handout": prior,
|
||||
"session_recap": newRecord,
|
||||
},
|
||||
Invocation: map[string]manifest.AnalyzeArtifactRecord{"session_recap": newRecord},
|
||||
}}, nil
|
||||
}}
|
||||
|
||||
_, err := executeStages(context.Background(), cfg, []stage.Stage{stageToRun}, RunOptions{
|
||||
Force: true,
|
||||
Env: &Env{ManifestStore: store},
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), "injected analyze projection save failure") {
|
||||
t.Fatalf("executeStages() error = %v", err)
|
||||
}
|
||||
if !store.failed {
|
||||
t.Fatal("projection persistence failure was not injected")
|
||||
}
|
||||
loaded, loadErr := store.delegate.Load(context.Background(), manifestPathFor(cfg))
|
||||
if loadErr != nil {
|
||||
t.Fatalf("Load() error = %v", loadErr)
|
||||
}
|
||||
analyze := loaded.Stages["analyze"]
|
||||
if analyze.Status != manifest.StatusFailed || len(analyze.AnalyzeArtifacts) != 1 || !reflect.DeepEqual(analyze.AnalyzeArtifacts["player_handout"], prior) {
|
||||
t.Fatalf("durable analyze state after rollback = %#v", analyze)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteStagesRejectsAnalyzeProjectionFromOtherStage(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
stageToRun := projectionStage{name: "prepare", run: func(_ *stage.Env, m *manifest.Manifest) (*stage.StageResult, error) {
|
||||
record := appAnalyzeRecord("session_recap", manifest.AnalyzeArtifactCurrent, m.RunID, time.Now().UTC())
|
||||
return &stage.StageResult{AnalyzeState: &stage.AnalyzeStateProjection{
|
||||
Session: map[string]manifest.AnalyzeArtifactRecord{"session_recap": record},
|
||||
}}, nil
|
||||
}}
|
||||
_, err := executeStages(context.Background(), cfg, []stage.Stage{stageToRun}, RunOptions{})
|
||||
if err == nil || !strings.Contains(err.Error(), "returned analyze-owned state projection") {
|
||||
t.Fatalf("executeStages() error = %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteStagesRejectsContradictoryAnalyzeResultWithError(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
stageToRun := projectionStage{name: "analyze", run: func(_ *stage.Env, m *manifest.Manifest) (*stage.StageResult, error) {
|
||||
record := appAnalyzeRecord("session_recap", manifest.AnalyzeArtifactCurrent, m.RunID, time.Now().UTC())
|
||||
return &stage.StageResult{
|
||||
Outputs: []artifacts.Ref{{Kind: "session_recap", RelativePath: "artifacts/session-recap.md"}},
|
||||
AnalyzeState: &stage.AnalyzeStateProjection{
|
||||
Session: map[string]manifest.AnalyzeArtifactRecord{"session_recap": record},
|
||||
Invocation: map[string]manifest.AnalyzeArtifactRecord{"session_recap": record},
|
||||
},
|
||||
}, errors.New("analysis failed")
|
||||
}}
|
||||
_, err := executeStages(context.Background(), cfg, []stage.Stage{stageToRun}, RunOptions{})
|
||||
if err == nil || !strings.Contains(err.Error(), "may contain only analyze-owned state projection") {
|
||||
t.Fatalf("executeStages() error = %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteStagesExposesSelectedForceDecisionToStage(t *testing.T) {
|
||||
for _, force := range []bool{false, true} {
|
||||
t.Run(strings.ToLower(strings.TrimSpace(map[bool]string{false: "ordinary", true: "forced"}[force])), func(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
captured := !force
|
||||
stageToRun := projectionStage{name: "prepare", run: func(env *stage.Env, _ *manifest.Manifest) (*stage.StageResult, error) {
|
||||
captured = env.Force
|
||||
return &stage.StageResult{}, nil
|
||||
}}
|
||||
if _, err := executeStages(context.Background(), cfg, []stage.Stage{stageToRun}, RunOptions{Force: force}); err != nil {
|
||||
t.Fatalf("executeStages() error = %v", err)
|
||||
}
|
||||
if captured != force {
|
||||
t.Fatalf("stage env force = %v, want %v", captured, force)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func appAnalyzeRecord(key string, status manifest.AnalyzeArtifactStatus, producerRunID string, at time.Time) manifest.AnalyzeArtifactRecord {
|
||||
record := manifest.AnalyzeArtifactRecord{
|
||||
Key: key,
|
||||
Status: status,
|
||||
ProducerRunID: producerRunID,
|
||||
UpdatedAt: at,
|
||||
}
|
||||
if status == manifest.AnalyzeArtifactFailed {
|
||||
record.Error = "scriptorium failed"
|
||||
return record
|
||||
}
|
||||
if status != manifest.AnalyzeArtifactCurrent {
|
||||
record.FingerprintVersion = manifest.AnalyzeFingerprintContractVersion
|
||||
record.Fingerprint = strings.Repeat("b", 64)
|
||||
return record
|
||||
}
|
||||
record.FingerprintVersion = manifest.AnalyzeFingerprintContractVersion
|
||||
record.Fingerprint = strings.Repeat("a", 64)
|
||||
record.OutputSize = 42
|
||||
record.Output = &manifest.ArtifactRecord{
|
||||
Kind: key,
|
||||
SourceID: artifacts.ConfiguredArtifactSourceID(key),
|
||||
LocalPath: "artifacts/" + strings.ReplaceAll(key, "_", "-") + ".md",
|
||||
ProducerRunID: producerRunID,
|
||||
Checksum: strings.Repeat("c", 64),
|
||||
Contract: &artifactmodel.ContractMetadata{
|
||||
MediaType: "text/markdown", SchemaID: "narratio." + key, SchemaVersion: "1",
|
||||
},
|
||||
}
|
||||
return record
|
||||
}
|
||||
|
||||
func analyzeArtifactOutputKeys(outputs []manifest.ArtifactRecord) []string {
|
||||
keys := make([]string, 0, len(outputs))
|
||||
for _, output := range outputs {
|
||||
keys = append(keys, strings.TrimPrefix(output.SourceID, "narratio.artifact."))
|
||||
}
|
||||
return keys
|
||||
}
|
||||
|
||||
type analyzeProjectionFailingStore struct {
|
||||
delegate *manifest.LocalStore
|
||||
shouldFail func(*manifest.Manifest) bool
|
||||
failed bool
|
||||
}
|
||||
|
||||
func (s *analyzeProjectionFailingStore) Create(ctx context.Context, sessionID string) (*manifest.Manifest, error) {
|
||||
return s.delegate.Create(ctx, sessionID)
|
||||
}
|
||||
|
||||
func (s *analyzeProjectionFailingStore) Load(ctx context.Context, path string) (*manifest.Manifest, error) {
|
||||
return s.delegate.Load(ctx, path)
|
||||
}
|
||||
|
||||
func (s *analyzeProjectionFailingStore) Save(ctx context.Context, path string, m *manifest.Manifest) error {
|
||||
if !s.failed && s.shouldFail != nil && s.shouldFail(m) {
|
||||
s.failed = true
|
||||
return errors.New("injected analyze projection save failure")
|
||||
}
|
||||
return s.delegate.Save(ctx, path, m)
|
||||
}
|
||||
295
internal/app/assembled_workflow_test.go
Normal file
295
internal/app/assembled_workflow_test.go
Normal file
@@ -0,0 +1,295 @@
|
||||
package app
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/scriptorium"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/stage"
|
||||
)
|
||||
|
||||
func TestAssembledFullRunUsesCanonicalOrderAndBoundedRunManifests(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
canonical := []string{
|
||||
"prepare", "transcribe", "merge", "polish", "normalize", "trim",
|
||||
"render", "extract", "analyze", "publish", "notify",
|
||||
}
|
||||
var order []string
|
||||
stages := make([]stage.Stage, 0, len(canonical))
|
||||
for _, name := range canonical {
|
||||
stages = append(stages, resultStage{name: name, result: &stage.StageResult{}, order: &order})
|
||||
}
|
||||
summary, err := executeStages(context.Background(), cfg, stages, RunOptions{})
|
||||
if err != nil {
|
||||
t.Fatalf("executeStages() error = %v", err)
|
||||
}
|
||||
if !reflect.DeepEqual(order, canonical) || !reflect.DeepEqual(summary.Executed, canonical) {
|
||||
t.Fatalf("execution order=%#v summary=%#v, want %#v", order, summary.Executed, canonical)
|
||||
}
|
||||
runManifest, err := (&manifest.LocalStore{}).LoadRun(context.Background(), summary.RunManifestPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !reflect.DeepEqual(runManifest.RequestedStages, canonical) {
|
||||
t.Fatalf("requested stages = %#v, want canonical order", runManifest.RequestedStages)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssembledCanonicalAndAliasArtifactRegenerationRequestsMatch(t *testing.T) {
|
||||
workspaceRoot := t.TempDir()
|
||||
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
|
||||
type capturedRequest struct {
|
||||
stages []string
|
||||
artifacts []string
|
||||
force bool
|
||||
}
|
||||
var captured []capturedRequest
|
||||
original := executeStagesFn
|
||||
t.Cleanup(func() { executeStagesFn = original })
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, plan BoundedPlan, options RunOptions) (*RunSummary, error) {
|
||||
captured = append(captured, capturedRequest{
|
||||
stages: plan.Names(), artifacts: append([]string(nil), options.SelectedArtifacts...), force: options.Force,
|
||||
})
|
||||
return &RunSummary{SessionID: "2026-05-03", ManifestPath: manifestPathForConfig(workspaceRoot)}, nil
|
||||
}
|
||||
base := []string{
|
||||
"2026-05-03", "--force", "--from", "extract", "--through", "analyze",
|
||||
"--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath,
|
||||
}
|
||||
if err := Run(context.Background(), base, &bytes.Buffer{}); err != nil {
|
||||
t.Fatalf("canonical unselected Run() error = %v", err)
|
||||
}
|
||||
selected := append(append([]string(nil), base...), "--artifacts", "session_recap")
|
||||
if err := Run(context.Background(), selected, &bytes.Buffer{}); err != nil {
|
||||
t.Fatalf("canonical selected Run() error = %v", err)
|
||||
}
|
||||
var stdout, stderr bytes.Buffer
|
||||
alias := []string{
|
||||
"regenerate-artifacts", "2026-05-03", "--artifacts", "session_recap",
|
||||
"--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath,
|
||||
}
|
||||
if code := Execute(alias, &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("alias exit=%d stderr=%q", code, stderr.String())
|
||||
}
|
||||
if len(captured) != 3 {
|
||||
t.Fatalf("captured requests = %#v", captured)
|
||||
}
|
||||
wantStages := []string{"extract", "analyze"}
|
||||
if !reflect.DeepEqual(captured[0].stages, wantStages) || len(captured[0].artifacts) != 0 || !captured[0].force {
|
||||
t.Fatalf("unselected request = %#v", captured[0])
|
||||
}
|
||||
if !reflect.DeepEqual(captured[1], captured[2]) || !reflect.DeepEqual(captured[1].stages, wantStages) ||
|
||||
!reflect.DeepEqual(captured[1].artifacts, []string{"session_recap"}) || !captured[1].force {
|
||||
t.Fatalf("canonical=%#v alias=%#v, want identical bounded request", captured[1], captured[2])
|
||||
}
|
||||
}
|
||||
|
||||
func TestAssembledForcedSiblingIndependenceAndFailureBoundary(t *testing.T) {
|
||||
for _, selected := range []string{"render", "extract"} {
|
||||
t.Run(selected, func(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
seedAllStagesSucceeded(t, cfg)
|
||||
plan := mustBoundedPlan(t, selected, selected)
|
||||
runs := 0
|
||||
plan.stages = []stage.Stage{countingStage{name: selected, runs: &runs}}
|
||||
summary, err := executePlan(context.Background(), cfg, plan, RunOptions{Force: true})
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
loaded := loadAssembledManifest(t, cfg)
|
||||
sibling := "render"
|
||||
if selected == "render" {
|
||||
sibling = "extract"
|
||||
}
|
||||
if loaded.Stages[sibling].Status != manifest.StatusSucceeded {
|
||||
t.Fatalf("%s sibling = %#v, want succeeded", sibling, loaded.Stages[sibling])
|
||||
}
|
||||
for _, dependent := range []string{"analyze", "publish", "notify"} {
|
||||
if loaded.Stages[dependent].Status != manifest.StatusStale {
|
||||
t.Fatalf("%s status = %q, want stale", dependent, loaded.Stages[dependent].Status)
|
||||
}
|
||||
}
|
||||
runManifest, err := (&manifest.LocalStore{}).LoadRun(context.Background(), summary.RunManifestPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !reflect.DeepEqual(runManifest.RequestedStages, []string{selected}) || len(runManifest.Stages) != 1 {
|
||||
t.Fatalf("bounded run manifest = %#v", runManifest)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
t.Run("stop on failure", func(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
seedAllStagesSucceeded(t, cfg)
|
||||
plan := mustBoundedPlan(t, "render", "render")
|
||||
plan.stages = []stage.Stage{failingStage{name: "render", err: context.Canceled}}
|
||||
_, err := executePlan(context.Background(), cfg, plan, RunOptions{Force: true})
|
||||
if err == nil {
|
||||
t.Fatal("forced render failure returned nil")
|
||||
}
|
||||
loaded := loadAssembledManifest(t, cfg)
|
||||
if loaded.Stages["extract"].Status != manifest.StatusSucceeded {
|
||||
t.Fatalf("extract sibling = %#v", loaded.Stages["extract"])
|
||||
}
|
||||
for _, outside := range []string{"analyze", "publish", "notify"} {
|
||||
if loaded.Stages[outside].Status != manifest.StatusStale {
|
||||
t.Fatalf("outside stage %s = %#v, want stale and unexecuted", outside, loaded.Stages[outside])
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestAssembledLegacyAnalyzeTransitionPublishesOnlyCurrentRecords(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
cfg.Pipeline.Scriptorium = &config.ScriptoriumConfig{Artifacts: map[string]config.ScriptoriumArtifactConfig{
|
||||
"player_handout": {Enabled: true, PromptID: "dnd.player_handout", OutputPath: "artifacts/player_handout.md"},
|
||||
"session_recap": {Enabled: true, PromptID: "dnd.session_recap", OutputPath: "artifacts/session_recap.md"},
|
||||
}}
|
||||
cfg.Pipeline.Storage.Backend = config.StorageBackendS3
|
||||
cfg.Pipeline.Storage.S3 = &config.StorageS3Config{Bucket: "archive", RootPrefix: "dnd"}
|
||||
cfg.Pipeline.Publish = &config.PublishConfig{
|
||||
Enabled: boolPtr(true), UploadRun: boolPtr(true),
|
||||
Outputs: []config.PublishOutputRule{
|
||||
{Source: artifacts.ConfiguredArtifactSourceID("player_handout"), Dest: "artifacts/player_handout.md", Required: boolPtr(true)},
|
||||
{Source: artifacts.ConfiguredArtifactSourceID("session_recap"), Dest: "artifacts/session_recap.md", Required: boolPtr(true)},
|
||||
},
|
||||
}
|
||||
paths, err := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root).EnsureLayoutFor(cfg.Session.Campaign, cfg.Session.SessionID)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
legacyHandout := []byte("legacy handout\n")
|
||||
legacyRecap := []byte("legacy recap\n")
|
||||
mustWriteTestFile(t, filepath.Join(paths.ArtifactsDir, "player_handout.md"), string(legacyHandout))
|
||||
mustWriteTestFile(t, filepath.Join(paths.ArtifactsDir, "session_recap.md"), string(legacyRecap))
|
||||
|
||||
now := time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC)
|
||||
m := manifest.New(cfg.Session.SessionID, now)
|
||||
m.Campaign = cfg.Session.Campaign
|
||||
for index, name := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim"} {
|
||||
m.MarkStageSucceeded(name, now.Add(time.Duration(index)*time.Minute), nil)
|
||||
}
|
||||
// Historical manifests can show extract completing before render. Status,
|
||||
// not the old relative timestamps, is the compatibility authority.
|
||||
m.MarkStageSucceeded("extract", now.Add(10*time.Minute), nil)
|
||||
m.MarkStageSucceeded("render", now.Add(11*time.Minute), nil)
|
||||
m.MarkStageSucceeded("analyze", now.Add(12*time.Minute), []manifest.ArtifactRecord{
|
||||
{Kind: "player_handout", LocalPath: "artifacts/player_handout.md"},
|
||||
{Kind: "session_recap", LocalPath: "artifacts/session_recap.md"},
|
||||
})
|
||||
if err := (&manifest.LocalStore{}).Save(context.Background(), paths.ManifestPath, m); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
analyze, err := stage.Select("analyze")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
fake := &scriptorium.FakeRunner{}
|
||||
_, err = executeStages(context.Background(), cfg, []stage.Stage{analyze}, RunOptions{
|
||||
SelectedArtifacts: []string{"session_recap"}, Env: &Env{Scriptorium: fake},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("partial legacy regeneration: %v", err)
|
||||
}
|
||||
if len(fake.RunRequests) != 1 || fake.RunRequests[0].PromptID != "dnd.session_recap" {
|
||||
t.Fatalf("partial requests = %#v", fake.RunRequests)
|
||||
}
|
||||
afterPartial := loadAssembledManifest(t, cfg)
|
||||
if len(afterPartial.Stages["analyze"].AnalyzeArtifacts) != 1 ||
|
||||
afterPartial.Stages["analyze"].AnalyzeArtifacts["session_recap"].Status != manifest.AnalyzeArtifactCurrent {
|
||||
t.Fatalf("partial state = %#v", afterPartial.Stages["analyze"].AnalyzeArtifacts)
|
||||
}
|
||||
for _, transcriptStage := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "render"} {
|
||||
if afterPartial.Stages[transcriptStage].Status != manifest.StatusSucceeded {
|
||||
t.Fatalf("legacy transition invalidated %s: %#v", transcriptStage, afterPartial.Stages[transcriptStage])
|
||||
}
|
||||
}
|
||||
configured := artifacts.ConfiguredArtifactDefinitions(cfg.Pipeline.Scriptorium.Artifacts)
|
||||
effective, err := artifacts.ResolveEffectiveArtifactSet(configured, nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
catalog, err := artifacts.BootstrapRuntimeCatalog(configured, effective, nil)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
catalog.HydrateAnalyzeArtifacts(paths, afterPartial, configured)
|
||||
if entry, ok := catalog.Lookup(artifacts.ConfiguredArtifactSourceID("player_handout")); !ok || entry.Available {
|
||||
t.Fatalf("legacy unselected handout catalog entry = %#v, present=%v", entry, ok)
|
||||
}
|
||||
|
||||
full, err := executeStages(context.Background(), cfg, []stage.Stage{analyze}, RunOptions{Env: &Env{Scriptorium: fake}})
|
||||
if err != nil {
|
||||
t.Fatalf("full regeneration: %v", err)
|
||||
}
|
||||
if len(fake.RunRequests) != 2 || fake.RunRequests[1].PromptID != "dnd.player_handout" {
|
||||
t.Fatalf("full requests = %#v, want only missing handout added", fake.RunRequests)
|
||||
}
|
||||
fullRun, err := (&manifest.LocalStore{}).LoadRun(context.Background(), full.RunManifestPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := analyzeArtifactOutputKeys(fullRun.Stages["analyze"].Outputs); !reflect.DeepEqual(got, []string{"player_handout"}) {
|
||||
t.Fatalf("full invocation outputs = %#v, want newly generated handout only", got)
|
||||
}
|
||||
afterFull := loadAssembledManifest(t, cfg)
|
||||
for _, key := range []string{"player_handout", "session_recap"} {
|
||||
if afterFull.Stages["analyze"].AnalyzeArtifacts[key].Status != manifest.AnalyzeArtifactCurrent {
|
||||
t.Fatalf("%s state = %#v", key, afterFull.Stages["analyze"].AnalyzeArtifacts[key])
|
||||
}
|
||||
}
|
||||
|
||||
publish, err := stage.Select("publish")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
remote := &storage.FakeBackend{}
|
||||
published, err := executeStages(context.Background(), cfg, []stage.Stage{publish}, RunOptions{Env: &Env{ObjectStore: remote}})
|
||||
if err != nil {
|
||||
t.Fatalf("publish current records: %v", err)
|
||||
}
|
||||
afterPublish := loadAssembledManifest(t, cfg)
|
||||
if got := afterPublish.Stages["publish"].Metadata["published_files_uploaded"]; got != float64(2) {
|
||||
t.Fatalf("published files = %#v, want 2", got)
|
||||
}
|
||||
publishRun, err := (&manifest.LocalStore{}).LoadRun(context.Background(), published.RunManifestPath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !reflect.DeepEqual(publishRun.RequestedStages, []string{"publish"}) || publishRun.Stages["analyze"] != nil {
|
||||
t.Fatalf("publish run manifest = %#v", publishRun)
|
||||
}
|
||||
}
|
||||
|
||||
func seedAllStagesSucceeded(t *testing.T, cfg *config.Config) {
|
||||
t.Helper()
|
||||
m := manifest.New(cfg.Session.SessionID, time.Now().UTC())
|
||||
m.Campaign = cfg.Session.Campaign
|
||||
for _, name := range canonicalStageNames() {
|
||||
m.MarkStageSucceeded(name, time.Now().UTC(), nil)
|
||||
}
|
||||
saveBoundedManifest(t, cfg, m)
|
||||
}
|
||||
|
||||
func loadAssembledManifest(t *testing.T, cfg *config.Config) *manifest.Manifest {
|
||||
t.Helper()
|
||||
m, err := (&manifest.LocalStore{}).Load(context.Background(), manifestPathFor(cfg))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
func manifestPathForConfig(workspaceRoot string) string {
|
||||
return artifacts.SessionManifestPathForCampaign(workspaceRoot, "sample-campaign", "2026-05-03")
|
||||
}
|
||||
62
internal/app/bounded_prerequisites.go
Normal file
62
internal/app/bounded_prerequisites.go
Normal file
@@ -0,0 +1,62 @@
|
||||
package app
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||
)
|
||||
|
||||
func validateBoundedPrerequisites(plan BoundedPlan, m *manifest.Manifest) error {
|
||||
if !plan.HasExplicitBounds() {
|
||||
return nil
|
||||
}
|
||||
for _, name := range plan.PrefixNames() {
|
||||
status := "absent"
|
||||
if m != nil && m.Stages != nil && m.Stages[name] != nil {
|
||||
stageStatus := m.Stages[name].Status
|
||||
if stageStatus == manifest.StatusSucceeded || stageStatus == manifest.StatusSkipped {
|
||||
continue
|
||||
}
|
||||
if stageStatus != "" {
|
||||
status = string(stageStatus)
|
||||
}
|
||||
}
|
||||
return fmt.Errorf(
|
||||
"prerequisite stage %q has unusable status %q before selected start %q; widen the range with --from %s or recover %s explicitly",
|
||||
name,
|
||||
status,
|
||||
plan.From(),
|
||||
name,
|
||||
name,
|
||||
)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func inspectBoundedPrerequisites(ctx context.Context, cfg *config.Config, plan BoundedPlan, store manifest.Store) error {
|
||||
if !plan.HasExplicitBounds() || len(plan.PrefixNames()) == 0 {
|
||||
return nil
|
||||
}
|
||||
if cfg == nil || cfg.Pipeline == nil || cfg.Session == nil {
|
||||
return fmt.Errorf("bounded prerequisite inspection requires resolved pipeline and session configuration")
|
||||
}
|
||||
if store == nil {
|
||||
store = &manifest.LocalStore{}
|
||||
}
|
||||
path := artifacts.SessionManifestPathForCampaign(
|
||||
cfg.Pipeline.Workspace.Root,
|
||||
cfg.Session.Campaign,
|
||||
cfg.Session.SessionID,
|
||||
)
|
||||
m, present, err := loadManifestAtPathIfPresent(ctx, store, path)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if !present {
|
||||
m = nil
|
||||
}
|
||||
return validateBoundedPrerequisites(plan, m)
|
||||
}
|
||||
345
internal/app/bounded_prerequisites_test.go
Normal file
345
internal/app/bounded_prerequisites_test.go
Normal file
@@ -0,0 +1,345 @@
|
||||
package app
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/stage"
|
||||
)
|
||||
|
||||
func TestValidateBoundedPrerequisitesRejectsFirstUnusablePrefixStatus(t *testing.T) {
|
||||
plan := mustBoundedPlan(t, "render", "extract")
|
||||
now := time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC)
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
status manifest.StageStatus
|
||||
want string
|
||||
}{
|
||||
{name: "absent", want: "absent"},
|
||||
{name: "pending", status: manifest.StatusPending, want: "pending"},
|
||||
{name: "running", status: manifest.StatusRunning, want: "running"},
|
||||
{name: "failed", status: manifest.StatusFailed, want: "failed"},
|
||||
{name: "stale", status: manifest.StatusStale, want: "stale"},
|
||||
{name: "interrupted", status: manifest.StatusInterrupted, want: "interrupted"},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
m := manifest.New("session", now)
|
||||
if test.status != "" {
|
||||
m.Stages["prepare"] = &manifest.StageRecord{Name: "prepare", Status: test.status}
|
||||
}
|
||||
// A later terminal prefix must not hide the first unusable one.
|
||||
m.MarkStageSucceeded("transcribe", now, nil)
|
||||
err := validateBoundedPrerequisites(plan, m)
|
||||
if err == nil {
|
||||
t.Fatal("validateBoundedPrerequisites() error = nil")
|
||||
}
|
||||
for _, detail := range []string{`stage "prepare"`, `status "` + test.want + `"`, `selected start "render"`, "--from prepare", "recover prepare"} {
|
||||
if !strings.Contains(err.Error(), detail) {
|
||||
t.Fatalf("error = %q, want detail %q", err, detail)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidateBoundedPrerequisitesAcceptsSucceededAndSkippedPrefix(t *testing.T) {
|
||||
plan := mustBoundedPlan(t, "render", "render")
|
||||
m := manifest.New("session", time.Now().UTC())
|
||||
for index, name := range plan.PrefixNames() {
|
||||
if index%2 == 0 {
|
||||
m.MarkStageSucceeded(name, time.Now().UTC(), nil)
|
||||
} else {
|
||||
m.MarkStageSkipped(name, time.Now().UTC(), "not applicable")
|
||||
}
|
||||
}
|
||||
if err := validateBoundedPrerequisites(plan, m); err != nil {
|
||||
t.Fatalf("validateBoundedPrerequisites() error = %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestValidateBoundedPrerequisitesHasNoPrefixAtPrepareAndIgnoresSuffix(t *testing.T) {
|
||||
preparePlan := mustBoundedPlan(t, "prepare", "prepare")
|
||||
if err := validateBoundedPrerequisites(preparePlan, nil); err != nil {
|
||||
t.Fatalf("prepare prerequisite validation error = %v", err)
|
||||
}
|
||||
|
||||
renderPlan := mustBoundedPlan(t, "render", "render")
|
||||
m := manifest.New("session", time.Now().UTC())
|
||||
markPrefixSucceeded(m, renderPlan)
|
||||
m.MarkStageFailed("analyze", time.Now().UTC(), "later failure")
|
||||
if err := validateBoundedPrerequisites(renderPlan, m); err != nil {
|
||||
t.Fatalf("suffix status affected prerequisite validation: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteStagesRejectsBoundedPrerequisitesBeforePersistentMutation(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
plan := mustBoundedPlan(t, "render", "render")
|
||||
m := manifest.New(cfg.Session.SessionID, time.Now().UTC())
|
||||
m.Campaign = cfg.Session.Campaign
|
||||
m.MarkStageRunning("prepare", time.Now().UTC())
|
||||
manifestPath := saveBoundedManifest(t, cfg, m)
|
||||
before, err := os.ReadFile(manifestPath)
|
||||
if err != nil {
|
||||
t.Fatalf("read seeded manifest: %v", err)
|
||||
}
|
||||
|
||||
store := &prerequisiteMutationSpy{local: &manifest.LocalStore{}}
|
||||
runs := 0
|
||||
plan.stages = []stage.Stage{countingStage{name: "render", runs: &runs}}
|
||||
_, err = executePlan(context.Background(), cfg, plan, RunOptions{
|
||||
Env: &Env{ManifestStore: store},
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), `stage "prepare" has unusable status "running"`) {
|
||||
t.Fatalf("executeStages() error = %v, want running prerequisite", err)
|
||||
}
|
||||
if runs != 0 || store.creates != 0 || store.saves != 0 {
|
||||
t.Fatalf("runs=%d manifest creates=%d saves=%d, want no mutation", runs, store.creates, store.saves)
|
||||
}
|
||||
after, err := os.ReadFile(manifestPath)
|
||||
if err != nil {
|
||||
t.Fatalf("read manifest after rejection: %v", err)
|
||||
}
|
||||
if string(after) != string(before) {
|
||||
t.Fatal("manifest changed after prerequisite rejection")
|
||||
}
|
||||
if _, err := os.Stat(artifacts.SessionRunsDirForCampaign(cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID)); !os.IsNotExist(err) {
|
||||
t.Fatalf("runs directory stat error = %v, want not exist", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteStagesRechecksBoundedPrerequisitesUnderSessionLock(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
plan := mustBoundedPlan(t, "render", "render")
|
||||
m := manifest.New(cfg.Session.SessionID, time.Now().UTC())
|
||||
m.Campaign = cfg.Session.Campaign
|
||||
markPrefixSucceeded(m, plan)
|
||||
saveBoundedManifest(t, cfg, m)
|
||||
|
||||
store := &prerequisiteChangingStore{local: &manifest.LocalStore{}}
|
||||
runs := 0
|
||||
plan.stages = []stage.Stage{countingStage{name: "render", runs: &runs}}
|
||||
_, err := executePlan(context.Background(), cfg, plan, RunOptions{
|
||||
Env: &Env{ManifestStore: store},
|
||||
})
|
||||
if err == nil || !strings.Contains(err.Error(), `stage "prepare" has unusable status "running"`) {
|
||||
t.Fatalf("executeStages() error = %v, want changed prerequisite rejection", err)
|
||||
}
|
||||
if store.loads != 2 {
|
||||
t.Fatalf("manifest loads = %d, want preflight and locked reload", store.loads)
|
||||
}
|
||||
if runs != 0 || store.creates != 0 || store.saves != 0 {
|
||||
t.Fatalf("runs=%d manifest creates=%d saves=%d, want no run or manifest mutation", runs, store.creates, store.saves)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteStagesBoundedCompositionUsesOnlySelectedCollaborators(t *testing.T) {
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
stageName string
|
||||
configure func(*config.Config)
|
||||
assertProbe func(*testing.T, *stage.Env)
|
||||
}{
|
||||
{
|
||||
name: "render",
|
||||
stageName: "render",
|
||||
assertProbe: func(t *testing.T, env *stage.Env) {
|
||||
if env.Seriatim == nil || env.Notarius != nil || env.Scriptorium != nil {
|
||||
t.Fatalf("render collaborators: seriatim=%v notarius=%v scriptorium=%v", env.Seriatim, env.Notarius, env.Scriptorium)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "extract",
|
||||
stageName: "extract",
|
||||
configure: func(cfg *config.Config) {
|
||||
cfg.Pipeline.Notarius = &config.NotariusConfig{Enabled: true}
|
||||
},
|
||||
assertProbe: func(t *testing.T, env *stage.Env) {
|
||||
if env.Notarius == nil || env.Scriptorium != nil || env.Seriatim != nil {
|
||||
t.Fatalf("extract collaborators: notarius=%v scriptorium=%v seriatim=%v", env.Notarius, env.Scriptorium, env.Seriatim)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "analyze",
|
||||
stageName: "analyze",
|
||||
assertProbe: func(t *testing.T, env *stage.Env) {
|
||||
if env.Scriptorium == nil || env.Notarius != nil || env.Seriatim != nil || env.WhisperX != nil || env.Audita != nil {
|
||||
t.Fatalf("analyze collaborators: scriptorium=%v notarius=%v seriatim=%v whisperx=%v audita=%v", env.Scriptorium, env.Notarius, env.Seriatim, env.WhisperX, env.Audita)
|
||||
}
|
||||
},
|
||||
},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
if test.configure != nil {
|
||||
test.configure(cfg)
|
||||
}
|
||||
plan := mustBoundedPlan(t, test.stageName, test.stageName)
|
||||
m := manifest.New(cfg.Session.SessionID, time.Now().UTC())
|
||||
m.Campaign = cfg.Session.Campaign
|
||||
markPrefixSucceeded(m, plan)
|
||||
saveBoundedManifest(t, cfg, m)
|
||||
|
||||
var captured *stage.Env
|
||||
plan.stages = []stage.Stage{collaboratorProbeStage{name: test.stageName, captured: &captured}}
|
||||
_, err := executePlan(context.Background(), cfg, plan, RunOptions{})
|
||||
if err != nil {
|
||||
t.Fatalf("executeStages() error = %v", err)
|
||||
}
|
||||
if captured == nil {
|
||||
t.Fatal("selected stage did not run")
|
||||
}
|
||||
test.assertProbe(t, captured)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteStagesBoundedForceStalesButDoesNotRunDependentsOutsideRange(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
plan := mustBoundedPlan(t, "render", "render")
|
||||
m := manifest.New(cfg.Session.SessionID, time.Now().UTC())
|
||||
m.Campaign = cfg.Session.Campaign
|
||||
for _, name := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim", "render", "extract", "analyze", "publish", "notify"} {
|
||||
m.MarkStageSucceeded(name, time.Now().UTC(), nil)
|
||||
}
|
||||
saveBoundedManifest(t, cfg, m)
|
||||
runs := 0
|
||||
plan.stages = []stage.Stage{countingStage{name: "render", runs: &runs}}
|
||||
_, err := executePlan(context.Background(), cfg, plan, RunOptions{Force: true})
|
||||
if err != nil {
|
||||
t.Fatalf("executeStages() error = %v", err)
|
||||
}
|
||||
if runs != 1 {
|
||||
t.Fatalf("selected render runs = %d, want 1", runs)
|
||||
}
|
||||
loaded, err := (&manifest.LocalStore{}).Load(context.Background(), manifestPathFor(cfg))
|
||||
if err != nil {
|
||||
t.Fatalf("load manifest: %v", err)
|
||||
}
|
||||
for _, name := range []string{"analyze", "publish", "notify"} {
|
||||
if loaded.Stages[name].Status != manifest.StatusStale {
|
||||
t.Fatalf("stage %q status = %q, want stale", name, loaded.Stages[name].Status)
|
||||
}
|
||||
}
|
||||
if loaded.Stages["extract"].Status != manifest.StatusSucceeded {
|
||||
t.Fatalf("extract status = %q, want succeeded", loaded.Stages["extract"].Status)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteStagesBoundedFailureStopsWithinSelectedRange(t *testing.T) {
|
||||
cfg := testConfig(t)
|
||||
plan := mustBoundedPlan(t, "render", "extract")
|
||||
m := manifest.New(cfg.Session.SessionID, time.Now().UTC())
|
||||
m.Campaign = cfg.Session.Campaign
|
||||
markPrefixSucceeded(m, plan)
|
||||
saveBoundedManifest(t, cfg, m)
|
||||
extractRuns := 0
|
||||
plan.stages = []stage.Stage{
|
||||
failingStage{name: "render", err: errors.New("render failed")},
|
||||
countingStage{name: "extract", runs: &extractRuns},
|
||||
}
|
||||
_, err := executePlan(context.Background(), cfg, plan, RunOptions{})
|
||||
if err == nil || !strings.Contains(err.Error(), "render failed") {
|
||||
t.Fatalf("executeStages() error = %v", err)
|
||||
}
|
||||
if extractRuns != 0 {
|
||||
t.Fatalf("extract runs = %d, want 0", extractRuns)
|
||||
}
|
||||
}
|
||||
|
||||
type collaboratorProbeStage struct {
|
||||
name string
|
||||
captured **stage.Env
|
||||
}
|
||||
|
||||
func (s collaboratorProbeStage) Name() string { return s.name }
|
||||
func (s collaboratorProbeStage) Run(_ context.Context, env *stage.Env, _ *manifest.Manifest) (*stage.StageResult, error) {
|
||||
*s.captured = env
|
||||
return &stage.StageResult{}, nil
|
||||
}
|
||||
|
||||
type prerequisiteMutationSpy struct {
|
||||
local *manifest.LocalStore
|
||||
creates int
|
||||
saves int
|
||||
}
|
||||
|
||||
type prerequisiteChangingStore struct {
|
||||
local *manifest.LocalStore
|
||||
loads int
|
||||
creates int
|
||||
saves int
|
||||
}
|
||||
|
||||
func (s *prerequisiteChangingStore) Create(ctx context.Context, sessionID string) (*manifest.Manifest, error) {
|
||||
s.creates++
|
||||
return s.local.Create(ctx, sessionID)
|
||||
}
|
||||
|
||||
func (s *prerequisiteChangingStore) Load(ctx context.Context, path string) (*manifest.Manifest, error) {
|
||||
s.loads++
|
||||
m, err := s.local.Load(ctx, path)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if s.loads == 2 {
|
||||
m.MarkStageRunning("prepare", time.Now().UTC())
|
||||
}
|
||||
return m, nil
|
||||
}
|
||||
|
||||
func (s *prerequisiteChangingStore) Save(ctx context.Context, path string, m *manifest.Manifest) error {
|
||||
s.saves++
|
||||
return s.local.Save(ctx, path, m)
|
||||
}
|
||||
|
||||
func (s *prerequisiteMutationSpy) Create(ctx context.Context, sessionID string) (*manifest.Manifest, error) {
|
||||
s.creates++
|
||||
return s.local.Create(ctx, sessionID)
|
||||
}
|
||||
|
||||
func (s *prerequisiteMutationSpy) Load(ctx context.Context, path string) (*manifest.Manifest, error) {
|
||||
return s.local.Load(ctx, path)
|
||||
}
|
||||
|
||||
func (s *prerequisiteMutationSpy) Save(ctx context.Context, path string, m *manifest.Manifest) error {
|
||||
s.saves++
|
||||
return s.local.Save(ctx, path, m)
|
||||
}
|
||||
|
||||
func mustBoundedPlan(t *testing.T, from, through string) BoundedPlan {
|
||||
t.Helper()
|
||||
plan, err := BuildBoundedPlan(from, through)
|
||||
if err != nil {
|
||||
t.Fatalf("BuildBoundedPlan(%q, %q) error = %v", from, through, err)
|
||||
}
|
||||
return plan
|
||||
}
|
||||
|
||||
func markPrefixSucceeded(m *manifest.Manifest, plan BoundedPlan) {
|
||||
for _, name := range plan.PrefixNames() {
|
||||
m.MarkStageSucceeded(name, time.Now().UTC(), nil)
|
||||
}
|
||||
}
|
||||
|
||||
func saveBoundedManifest(t *testing.T, cfg *config.Config, m *manifest.Manifest) string {
|
||||
t.Helper()
|
||||
path := manifestPathFor(cfg)
|
||||
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {
|
||||
t.Fatalf("create manifest directory: %v", err)
|
||||
}
|
||||
if err := (&manifest.LocalStore{}).Save(context.Background(), path, m); err != nil {
|
||||
t.Fatalf("save manifest: %v", err)
|
||||
}
|
||||
return path
|
||||
}
|
||||
110
internal/app/bounded_run.go
Normal file
110
internal/app/bounded_run.go
Normal file
@@ -0,0 +1,110 @@
|
||||
package app
|
||||
|
||||
import (
|
||||
"flag"
|
||||
"fmt"
|
||||
"io"
|
||||
"strconv"
|
||||
)
|
||||
|
||||
type boundedRunRequest struct {
|
||||
Config commonConfigFlags
|
||||
Plan BoundedPlan
|
||||
Force bool
|
||||
SelectedArtifacts []string
|
||||
}
|
||||
|
||||
type singletonStringFlag struct {
|
||||
name string
|
||||
value string
|
||||
set bool
|
||||
}
|
||||
|
||||
func (f *singletonStringFlag) String() string { return f.value }
|
||||
|
||||
func (f *singletonStringFlag) Set(value string) error {
|
||||
if f.set {
|
||||
return fmt.Errorf("--%s may be specified only once", f.name)
|
||||
}
|
||||
f.value = value
|
||||
f.set = true
|
||||
return nil
|
||||
}
|
||||
|
||||
type singletonBoolFlag struct {
|
||||
name string
|
||||
value bool
|
||||
set bool
|
||||
}
|
||||
|
||||
func (f *singletonBoolFlag) String() string { return strconv.FormatBool(f.value) }
|
||||
func (f *singletonBoolFlag) IsBoolFlag() bool { return true }
|
||||
|
||||
func (f *singletonBoolFlag) Set(raw string) error {
|
||||
if f.set {
|
||||
return fmt.Errorf("--%s may be specified only once", f.name)
|
||||
}
|
||||
value, err := strconv.ParseBool(raw)
|
||||
if err != nil {
|
||||
return fmt.Errorf("--%s requires a boolean value: %w", f.name, err)
|
||||
}
|
||||
f.value = value
|
||||
f.set = true
|
||||
return nil
|
||||
}
|
||||
|
||||
func parseBoundedRunRequest(command string, args []string, help io.Writer) (boundedRunRequest, error) {
|
||||
fs := flag.NewFlagSet(command, flag.ContinueOnError)
|
||||
fs.SetOutput(help)
|
||||
|
||||
var configFlags commonConfigFlags
|
||||
var from singletonStringFlag
|
||||
var through singletonStringFlag
|
||||
var force singletonBoolFlag
|
||||
var selectedArtifacts artifactSelectionFlag
|
||||
from.name = "from"
|
||||
through.name = "through"
|
||||
force.name = "force"
|
||||
|
||||
addCommonConfigFlags(fs, &configFlags)
|
||||
fs.Var(&from, "from", "first canonical stage to select (inclusive)")
|
||||
fs.Var(&through, "through", "last canonical stage to select (inclusive)")
|
||||
fs.Var(&force, "force", "rerun selected stages even when already succeeded")
|
||||
fs.Var(&selectedArtifacts, "artifacts", "configured artifact names to execute and publish (comma-separated or repeatable)")
|
||||
fs.Usage = func() {
|
||||
invocation := "narratio run"
|
||||
if command == "plan" {
|
||||
invocation = "narratio session plan"
|
||||
}
|
||||
_, _ = fmt.Fprintf(help, "Usage: %s <session_id> [--from <stage>] [--through <stage>] [--force] [--artifacts <name[,name...]>] [common config flags]\n\n", invocation)
|
||||
_, _ = fmt.Fprintln(help, "Bounds are inclusive; omitted --from or --through selects the beginning or end of the canonical pipeline.")
|
||||
_, _ = fmt.Fprintln(help)
|
||||
_, _ = fmt.Fprintln(help, "Flags:")
|
||||
fs.PrintDefaults()
|
||||
}
|
||||
|
||||
if err := parseSessionAwareFlags(command, fs, args, &configFlags.sessionID); err != nil {
|
||||
return boundedRunRequest{}, err
|
||||
}
|
||||
if configFlags.sessionID == "" {
|
||||
return boundedRunRequest{}, fmt.Errorf("%s: session_id is required", command)
|
||||
}
|
||||
plan, err := BuildBoundedPlan(from.value, through.value)
|
||||
if err != nil {
|
||||
return boundedRunRequest{}, fmt.Errorf("%s: %w", command, err)
|
||||
}
|
||||
normalizedArtifacts, err := selectedArtifacts.Normalize()
|
||||
if err != nil {
|
||||
return boundedRunRequest{}, fmt.Errorf("%s: invalid --artifacts: %w", command, err)
|
||||
}
|
||||
if len(normalizedArtifacts) > 0 && !plan.Contains("analyze") && !plan.Contains("publish") {
|
||||
return boundedRunRequest{}, fmt.Errorf("%s: --artifacts requires a selected range containing analyze or publish", command)
|
||||
}
|
||||
|
||||
return boundedRunRequest{
|
||||
Config: configFlags,
|
||||
Plan: plan,
|
||||
Force: force.value,
|
||||
SelectedArtifacts: normalizedArtifacts,
|
||||
}, nil
|
||||
}
|
||||
197
internal/app/bounded_run_test.go
Normal file
197
internal/app/bounded_run_test.go
Normal file
@@ -0,0 +1,197 @@
|
||||
package app
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"io"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||
)
|
||||
|
||||
func TestBoundedRunParsingIsSharedByRunAndPlan(t *testing.T) {
|
||||
args := []string{
|
||||
"2026-05-03",
|
||||
"--from", "extract",
|
||||
"--through=publish",
|
||||
"--force",
|
||||
"--artifacts", "session_recap,player_handout",
|
||||
"--artifacts=session_recap",
|
||||
"--config", "pipeline.yml",
|
||||
}
|
||||
runRequest, err := parseBoundedRunRequest("run", args, io.Discard)
|
||||
if err != nil {
|
||||
t.Fatalf("parse run request: %v", err)
|
||||
}
|
||||
planRequest, err := parseBoundedRunRequest("plan", args, io.Discard)
|
||||
if err != nil {
|
||||
t.Fatalf("parse plan request: %v", err)
|
||||
}
|
||||
if !reflect.DeepEqual(runRequest.Plan.Names(), planRequest.Plan.Names()) ||
|
||||
runRequest.Plan.From() != planRequest.Plan.From() ||
|
||||
runRequest.Plan.Through() != planRequest.Plan.Through() ||
|
||||
runRequest.Force != planRequest.Force ||
|
||||
!reflect.DeepEqual(runRequest.SelectedArtifacts, planRequest.SelectedArtifacts) ||
|
||||
runRequest.Config != planRequest.Config {
|
||||
t.Fatalf("run request = %#v, plan request = %#v", runRequest, planRequest)
|
||||
}
|
||||
wantArtifacts := []string{"player_handout", "session_recap"}
|
||||
if !reflect.DeepEqual(runRequest.SelectedArtifacts, wantArtifacts) {
|
||||
t.Fatalf("artifacts = %#v, want %#v", runRequest.SelectedArtifacts, wantArtifacts)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBoundedRunParsingRejectsDuplicateSingletons(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
args []string
|
||||
want string
|
||||
}{
|
||||
{name: "from separate", args: []string{"session", "--from", "render", "--from", "extract"}, want: "--from may be specified only once"},
|
||||
{name: "from equals", args: []string{"session", "--from=render", "--from=extract"}, want: "--from may be specified only once"},
|
||||
{name: "through mixed", args: []string{"session", "--through", "analyze", "--through=publish"}, want: "--through may be specified only once"},
|
||||
{name: "force separate", args: []string{"session", "--force", "--force"}, want: "--force may be specified only once"},
|
||||
{name: "force equals", args: []string{"session", "--force=true", "--force=false"}, want: "--force may be specified only once"},
|
||||
}
|
||||
for _, test := range tests {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
_, err := parseBoundedRunRequest("run", test.args, io.Discard)
|
||||
if err == nil || !strings.Contains(err.Error(), test.want) {
|
||||
t.Fatalf("error = %v, want %q", err, test.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestBoundedRunParsingUsesSharedRangeValidation(t *testing.T) {
|
||||
args := []string{"session", "--from", "publish", "--through", "render"}
|
||||
runRequest, runErr := parseBoundedRunRequest("run", args, io.Discard)
|
||||
planRequest, planErr := parseBoundedRunRequest("plan", args, io.Discard)
|
||||
if runErr == nil || planErr == nil {
|
||||
t.Fatalf("run request=%#v error=%v; plan request=%#v error=%v", runRequest, runErr, planRequest, planErr)
|
||||
}
|
||||
runDetail := strings.TrimPrefix(runErr.Error(), "run: ")
|
||||
planDetail := strings.TrimPrefix(planErr.Error(), "plan: ")
|
||||
if runDetail != planDetail || !strings.Contains(runDetail, `from stage "publish" occurs after through stage "render"`) {
|
||||
t.Fatalf("run error = %q, plan error = %q", runErr, planErr)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBoundedRunParsingGatesArtifactSelectionByRange(t *testing.T) {
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
through string
|
||||
wantErr bool
|
||||
}{
|
||||
{name: "render only", through: "render", wantErr: true},
|
||||
{name: "analyze only", through: "analyze"},
|
||||
{name: "publish only", through: "publish"},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
_, err := parseBoundedRunRequest("run", []string{
|
||||
"session", "--from", test.through, "--through", test.through, "--artifacts", "session_recap",
|
||||
}, io.Discard)
|
||||
if test.wantErr && (err == nil || !strings.Contains(err.Error(), "range containing analyze or publish")) {
|
||||
t.Fatalf("error = %v, want artifact/range error", err)
|
||||
}
|
||||
if !test.wantErr && err != nil {
|
||||
t.Fatalf("error = %v", err)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRunPassesBoundedPlanToRunner(t *testing.T) {
|
||||
workspaceRoot := t.TempDir()
|
||||
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
|
||||
|
||||
var capturedStages []string
|
||||
var capturedPlan BoundedPlan
|
||||
var capturedOptions RunOptions
|
||||
original := executeStagesFn
|
||||
t.Cleanup(func() { executeStagesFn = original })
|
||||
executeStagesFn = func(_ context.Context, _ *config.Config, plan BoundedPlan, options RunOptions) (*RunSummary, error) {
|
||||
capturedStages = plan.Names()
|
||||
capturedPlan = plan
|
||||
capturedOptions = options
|
||||
return &RunSummary{SessionID: "2026-05-03", ManifestPath: filepath.Join(workspaceRoot, "manifest.json")}, nil
|
||||
}
|
||||
|
||||
var out bytes.Buffer
|
||||
err := Run(context.Background(), []string{
|
||||
"2026-05-03",
|
||||
"--from", "render",
|
||||
"--through", "extract",
|
||||
"--force",
|
||||
"--config", pipelinePath,
|
||||
"--campaign-file", campaignPath,
|
||||
"--session", sessionPath,
|
||||
}, &out)
|
||||
if err != nil {
|
||||
t.Fatalf("Run() error = %v", err)
|
||||
}
|
||||
want := []string{"render", "extract"}
|
||||
if !reflect.DeepEqual(capturedStages, want) || !reflect.DeepEqual(capturedPlan.Names(), want) || !capturedOptions.Force {
|
||||
t.Fatalf("stages = %#v options = %#v", capturedStages, capturedOptions)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPlanPrintsOnlyBoundedRange(t *testing.T) {
|
||||
workspaceRoot := t.TempDir()
|
||||
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
|
||||
m := manifest.New("2026-05-03", time.Date(2026, 5, 3, 10, 0, 0, 0, time.UTC))
|
||||
m.Campaign = "sample-campaign"
|
||||
for _, name := range []string{"prepare", "transcribe", "merge", "polish", "normalize", "trim"} {
|
||||
m.MarkStageSucceeded(name, time.Date(2026, 5, 3, 10, 1, 0, 0, time.UTC), nil)
|
||||
}
|
||||
manifestPath := filepath.Join(workspaceRoot, "work", "sample-campaign", "2026-05-03", "manifest.json")
|
||||
if err := (&manifest.LocalStore{}).Save(context.Background(), manifestPath, m); err != nil {
|
||||
t.Fatalf("save prerequisite manifest: %v", err)
|
||||
}
|
||||
var out bytes.Buffer
|
||||
err := Plan(context.Background(), []string{
|
||||
"2026-05-03",
|
||||
"--from", "render",
|
||||
"--through", "extract",
|
||||
"--config", pipelinePath,
|
||||
"--campaign-file", campaignPath,
|
||||
"--session", sessionPath,
|
||||
}, &out)
|
||||
if err != nil {
|
||||
t.Fatalf("Plan() error = %v", err)
|
||||
}
|
||||
got := out.String()
|
||||
if !strings.Contains(got, "render: run\nextract: run\ntotals: run=2 skip=0") {
|
||||
t.Fatalf("output = %q, want bounded decisions", got)
|
||||
}
|
||||
if strings.Contains(got, "trim: ") || strings.Contains(got, "analyze: ") {
|
||||
t.Fatalf("output = %q, contains excluded stages", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBoundedRunCommandHelp(t *testing.T) {
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
args []string
|
||||
want string
|
||||
}{
|
||||
{name: "run", args: []string{"run", "--help"}, want: "Usage: narratio run <session_id> [--from <stage>] [--through <stage>]"},
|
||||
{name: "plan", args: []string{"session", "plan", "--help"}, want: "Usage: narratio session plan <session_id> [--from <stage>] [--through <stage>]"},
|
||||
} {
|
||||
t.Run(test.name, func(t *testing.T) {
|
||||
var stdout bytes.Buffer
|
||||
var stderr bytes.Buffer
|
||||
if code := Execute(test.args, &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("exit code = %d, stderr = %q", code, stderr.String())
|
||||
}
|
||||
if !strings.Contains(stdout.String(), test.want) || stderr.Len() != 0 {
|
||||
t.Fatalf("stdout = %q stderr = %q, want %q", stdout.String(), stderr.String(), test.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -6,6 +6,7 @@ import (
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/pathsafe"
|
||||
)
|
||||
|
||||
func resolveCampaignConfigPath(pipelineCfg *config.PipelineConfig, campaignIDFlag, campaignFileFlag string) (string, error) {
|
||||
@@ -33,12 +34,8 @@ func resolveCampaignConfigPath(pipelineCfg *config.PipelineConfig, campaignIDFla
|
||||
}
|
||||
|
||||
func validateCampaignIDToken(campaignID string) error {
|
||||
if filepath.IsAbs(campaignID) ||
|
||||
strings.Contains(campaignID, "/") ||
|
||||
strings.Contains(campaignID, `\`) ||
|
||||
campaignID == "." ||
|
||||
campaignID == ".." {
|
||||
return fmt.Errorf("campaign id %q must be a single path segment", campaignID)
|
||||
if err := pathsafe.ValidateOpaqueSegment(campaignID); err != nil {
|
||||
return fmt.Errorf("campaign id %q must be a single path segment and opaque identifier: %w", campaignID, err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -11,6 +11,7 @@ import (
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
// Clean removes local workspace/spool state while preserving durable cache
|
||||
@@ -39,10 +40,12 @@ func cleanSession(ctx context.Context, flags commonConfigFlags, dryRun, clearCac
|
||||
if strings.TrimSpace(flags.sessionID) == "" {
|
||||
return fmt.Errorf("clean: session_id is required unless --all is set")
|
||||
}
|
||||
cfg, err := loadCommandConfig(ctx, flags.pipelinePath, flags.campaignPath, flags.campaignFilePath, flags.sessionPath, flags.sessionOptions())
|
||||
loaded, err := loadCommandConfig(ctx, flags.pipelinePath, flags.campaignPath, flags.campaignFilePath, flags.sessionPath, flags.sessionOptions())
|
||||
if err != nil {
|
||||
return fmt.Errorf("clean: %w", err)
|
||||
}
|
||||
defer func() { _ = loaded.Close() }()
|
||||
cfg := loaded.Config
|
||||
if cfg == nil || cfg.Pipeline == nil || cfg.Session == nil {
|
||||
return fmt.Errorf("clean: resolved pipeline and session config are required")
|
||||
}
|
||||
@@ -135,7 +138,7 @@ func reportCleanScopedDir(out io.Writer, root, target, policy string, dryRun boo
|
||||
fmt.Fprintf(out, "Missing: %s\n", dir.TargetAbs)
|
||||
return nil
|
||||
}
|
||||
if err := os.RemoveAll(dir.TargetAbs); err != nil {
|
||||
if err := fileops.RemoveAllUnderRoot(dir.RootAbs, dir.TargetAbs); err != nil {
|
||||
return fmt.Errorf("cleanup policy %s: remove %q: %w", policy, dir.TargetAbs, err)
|
||||
}
|
||||
fmt.Fprintf(out, "Deleted: %s\n", dir.TargetAbs)
|
||||
@@ -160,7 +163,7 @@ func reportCleanRootChildren(out io.Writer, root, policy string, dryRun bool) er
|
||||
fmt.Fprintf(out, "Would delete: %s\n", entry)
|
||||
continue
|
||||
}
|
||||
if err := os.RemoveAll(entry); err != nil {
|
||||
if err := fileops.RemoveAllUnderRoot(rootAbs, entry); err != nil {
|
||||
return fmt.Errorf("cleanup policy %s: remove %q: %w", policy, entry, err)
|
||||
}
|
||||
fmt.Fprintf(out, "Deleted: %s\n", entry)
|
||||
@@ -265,7 +268,7 @@ func reportCleanScopedFile(out io.Writer, root, target, policy string, dryRun bo
|
||||
fmt.Fprintf(out, "Missing cache file: %s\n", file.TargetAbs)
|
||||
return false, nil
|
||||
}
|
||||
if err := os.Remove(file.TargetAbs); err != nil {
|
||||
if err := fileops.RemoveAllUnderRoot(file.RootAbs, file.TargetAbs); err != nil {
|
||||
return false, fmt.Errorf("cleanup policy %s: remove %q: %w", policy, file.TargetAbs, err)
|
||||
}
|
||||
fmt.Fprintf(out, "Deleted cache file: %s\n", file.TargetAbs)
|
||||
|
||||
@@ -7,7 +7,9 @@ import (
|
||||
"strings"
|
||||
)
|
||||
|
||||
var supportedCommands = []string{"run", "run-stage", "analyze", "publish", "clean", "session"}
|
||||
var supportedCommands = []string{"version", "run", "regenerate-artifacts", "run-stage", "analyze", "publish", "clean", "session"}
|
||||
|
||||
var runCommandFn = Run
|
||||
|
||||
// Execute dispatches CLI commands and returns a process exit code.
|
||||
func Execute(args []string, stdout, stderr io.Writer) int {
|
||||
@@ -22,8 +24,12 @@ func Execute(args []string, stdout, stderr io.Writer) int {
|
||||
|
||||
var err error
|
||||
switch cmd {
|
||||
case "version":
|
||||
err = Version(cmdArgs, stdout)
|
||||
case "run":
|
||||
err = Run(ctx, cmdArgs, stdout)
|
||||
err = runCommandFn(ctx, cmdArgs, stdout)
|
||||
case "regenerate-artifacts":
|
||||
err = RegenerateArtifacts(ctx, cmdArgs, stdout)
|
||||
case "run-stage":
|
||||
err = RunStage(ctx, cmdArgs, stdout)
|
||||
case "analyze":
|
||||
|
||||
@@ -32,7 +32,7 @@ func TestExecuteValidCommands(t *testing.T) {
|
||||
wantOut string
|
||||
}{
|
||||
{name: "run", args: []string{"run", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "narratio run: session 2026-05-03; executed=11 skipped=1; manifest="},
|
||||
{name: "session plan", args: []string{"session", "plan", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "prepare: skip\ntranscribe: skip\nmerge: skip\npolish: skip\nnormalize: skip\ntrim: skip\nextract: run\nrender: skip\nanalyze: skip\npublish: skip\nnotify: skip"},
|
||||
{name: "session plan", args: []string{"session", "plan", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "analyze: skip\n targets: none\n prerequisites: none\n execute: none\n reuse: none\npublish: skip\nnotify: skip"},
|
||||
{name: "session status", args: []string{"session", "status", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "Session: 2026-05-03"},
|
||||
{name: "run-stage", args: []string{"run-stage", "polish", "2026-05-03", "--config", pipelinePath, "--campaign-file", campaignPath, "--session", sessionPath}, wantOut: "narratio run-stage: stage=polish executed=0 skipped=1 force=false; manifest="},
|
||||
}
|
||||
@@ -189,7 +189,7 @@ func TestExecuteRunStagePolishLoadsCredentialFromSecretsDir(t *testing.T) {
|
||||
configDir := t.TempDir()
|
||||
sessionID := "2026-05-03"
|
||||
secretsDir := filepath.Join(configDir, "secrets")
|
||||
if err := os.MkdirAll(secretsDir, 0o755); err != nil {
|
||||
if err := os.MkdirAll(secretsDir, secretDirectoryPrivateMode); err != nil {
|
||||
t.Fatalf("MkdirAll(%q): %v", secretsDir, err)
|
||||
}
|
||||
if err := os.WriteFile(filepath.Join(secretsDir, "OPENROUTER_API_KEY"), []byte("from-secret-file\n"), 0o600); err != nil {
|
||||
@@ -218,7 +218,7 @@ audita:
|
||||
binary: ` + auditaBinary + `
|
||||
llm_api_key_env: OPENROUTER_API_KEY
|
||||
notification:
|
||||
timeout: 10s
|
||||
mode: noop
|
||||
`
|
||||
sessionYAML := `session_id: ` + sessionID + `
|
||||
campaign: sample-campaign
|
||||
@@ -283,7 +283,7 @@ seriatim:
|
||||
audita:
|
||||
binary: audita
|
||||
notification:
|
||||
timeout: 10s
|
||||
mode: noop
|
||||
`
|
||||
sessionYAML := `session_id: 2026-05-03
|
||||
campaign: sample-campaign
|
||||
@@ -308,8 +308,8 @@ inputs:
|
||||
if code == 0 {
|
||||
t.Fatal("exit code = 0, want non-zero")
|
||||
}
|
||||
if !strings.Contains(stderr.String(), "read secrets env_dir") {
|
||||
t.Fatalf("stderr = %q, want secrets read-dir error context", stderr.String())
|
||||
if !strings.Contains(stderr.String(), "validate secrets env_dir") {
|
||||
t.Fatalf("stderr = %q, want secrets validation error context", stderr.String())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -476,9 +476,6 @@ func writeValidConfigFiles(t *testing.T, workspaceRoot string, transcribeURL ...
|
||||
seriatimBinary := writeSeriatimAppTestWrapper(t)
|
||||
scriptoriumBinary := writeScriptoriumAppTestWrapper(t)
|
||||
auditaBinary := writeAuditaAppTestWrapper(t)
|
||||
t.Setenv("GO_WANT_APP_SERIATIM_HELPER", "1")
|
||||
t.Setenv("GO_WANT_APP_SCRIPTORIUM_HELPER", "1")
|
||||
t.Setenv("GO_WANT_APP_AUDITA_HELPER", "1")
|
||||
t.Setenv("AUDITA_LLM_API_KEY", "test-audita-key")
|
||||
t.Setenv("PATH", filepath.Dir(scriptoriumBinary)+string(os.PathListSeparator)+os.Getenv("PATH"))
|
||||
|
||||
@@ -513,7 +510,7 @@ seriatim:
|
||||
audita:
|
||||
binary: ` + auditaBinary + `
|
||||
notification:
|
||||
timeout: 10s
|
||||
mode: noop
|
||||
`
|
||||
|
||||
sessionYAML := `session_id: 2026-05-03
|
||||
@@ -623,7 +620,7 @@ func writeScriptoriumAppTestWrapper(t *testing.T) string {
|
||||
}
|
||||
|
||||
func TestScriptoriumAppHelper(t *testing.T) {
|
||||
if os.Getenv("GO_WANT_APP_SCRIPTORIUM_HELPER") != "1" {
|
||||
if !appHelperInvocation() {
|
||||
return
|
||||
}
|
||||
|
||||
@@ -663,7 +660,7 @@ func TestScriptoriumAppHelper(t *testing.T) {
|
||||
}
|
||||
|
||||
func TestSeriatimAppHelper(t *testing.T) {
|
||||
if os.Getenv("GO_WANT_APP_SERIATIM_HELPER") != "1" {
|
||||
if !appHelperInvocation() {
|
||||
return
|
||||
}
|
||||
|
||||
@@ -725,7 +722,7 @@ func writeAuditaAppTestWrapper(t *testing.T) string {
|
||||
}
|
||||
|
||||
func TestAuditaAppHelper(t *testing.T) {
|
||||
if os.Getenv("GO_WANT_APP_AUDITA_HELPER") != "1" {
|
||||
if !appHelperInvocation() {
|
||||
return
|
||||
}
|
||||
|
||||
@@ -779,6 +776,15 @@ func TestAuditaAppHelper(t *testing.T) {
|
||||
os.Exit(0)
|
||||
}
|
||||
|
||||
func appHelperInvocation() bool {
|
||||
for _, arg := range os.Args {
|
||||
if arg == "--" {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func appSeriatimFlagValue(args []string, name string) string {
|
||||
for i := 0; i < len(args)-1; i++ {
|
||||
if args[i] == name {
|
||||
|
||||
@@ -2,13 +2,16 @@ package app
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/fileops"
|
||||
)
|
||||
|
||||
type pipelineCampaignConfig struct {
|
||||
@@ -18,14 +21,48 @@ type pipelineCampaignConfig struct {
|
||||
Campaign *config.CampaignConfig
|
||||
}
|
||||
|
||||
func loadCommandConfig(ctx context.Context, pipelineFlag, campaignFlag, campaignFileFlag, sessionFlag string, sessionOpts config.SessionLoadOptions) (*config.Config, error) {
|
||||
var downloadObjectToTempFn = storage.DownloadObjectToTemp
|
||||
|
||||
type commandConfig struct {
|
||||
Config *config.Config
|
||||
cleanup func() error
|
||||
}
|
||||
|
||||
func (c *commandConfig) Close() error {
|
||||
if c == nil || c.cleanup == nil {
|
||||
return nil
|
||||
}
|
||||
cleanup := c.cleanup
|
||||
c.cleanup = nil
|
||||
return cleanup()
|
||||
}
|
||||
|
||||
func retainedCommandConfig(cfg *config.Config) *commandConfig {
|
||||
return &commandConfig{Config: cfg}
|
||||
}
|
||||
|
||||
func loadCommandConfig(ctx context.Context, pipelineFlag, campaignFlag, campaignFileFlag, sessionFlag string, sessionOpts config.SessionLoadOptions) (loaded *commandConfig, err error) {
|
||||
var cleanup func() error
|
||||
defer func() {
|
||||
if err == nil || cleanup == nil {
|
||||
return
|
||||
}
|
||||
if cleanupErr := cleanup(); cleanupErr != nil {
|
||||
err = errors.Join(err, cleanupErr)
|
||||
}
|
||||
}()
|
||||
|
||||
base, err := loadPipelineCampaignConfig(pipelineFlag, campaignFlag, campaignFileFlag)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
if explicitSession := strings.TrimSpace(sessionFlag); explicitSession != "" {
|
||||
return config.LoadWithSessionOptions(base.PipelinePath, base.CampaignPath, explicitSession, sessionOpts)
|
||||
cfg, err := config.LoadWithSessionOptions(base.PipelinePath, base.CampaignPath, explicitSession, sessionOpts)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return retainedCommandConfig(cfg), nil
|
||||
}
|
||||
|
||||
discoveredSession, err := discoverSessionConfigPathWithCandidates(config.DefaultSessionConfigSearchPaths)
|
||||
@@ -33,7 +70,11 @@ func loadCommandConfig(ctx context.Context, pipelineFlag, campaignFlag, campaign
|
||||
return nil, err
|
||||
}
|
||||
if discoveredSession.Path != "" {
|
||||
return config.LoadWithSessionOptions(base.PipelinePath, base.CampaignPath, discoveredSession.Path, sessionOpts)
|
||||
cfg, err := config.LoadWithSessionOptions(base.PipelinePath, base.CampaignPath, discoveredSession.Path, sessionOpts)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return retainedCommandConfig(cfg), nil
|
||||
}
|
||||
|
||||
sessionID := strings.TrimSpace(sessionOpts.SessionID)
|
||||
@@ -41,7 +82,11 @@ func loadCommandConfig(ctx context.Context, pipelineFlag, campaignFlag, campaign
|
||||
return nil, missingSessionConfigError(discoveredSession.Searched, "remote session loading requires a session_id")
|
||||
}
|
||||
|
||||
sessionPrefix := artifacts.S3SessionPrefix(base.Pipeline.Storage.S3.RootPrefix, config.CampaignID(base.Campaign), sessionID)
|
||||
rootPrefix := ""
|
||||
if base.Pipeline.Storage.S3 != nil {
|
||||
rootPrefix = base.Pipeline.Storage.S3.RootPrefix
|
||||
}
|
||||
sessionPrefix := artifacts.S3SessionPrefix(rootPrefix, config.CampaignID(base.Campaign), sessionID)
|
||||
remoteKey := artifacts.S3SessionConfigKey(sessionPrefix)
|
||||
partialCfg := &config.Config{
|
||||
Pipeline: base.Pipeline,
|
||||
@@ -58,20 +103,26 @@ func loadCommandConfig(ctx context.Context, pipelineFlag, campaignFlag, campaign
|
||||
if err != nil {
|
||||
return nil, missingSessionConfigError(discoveredSession.Searched, err.Error())
|
||||
}
|
||||
sessionTempPath, err := storage.DownloadObjectToTemp(ctx, store, remoteKey, "narratio-session-*.yml")
|
||||
sessionTempPath, err := downloadObjectToTempFn(ctx, store, remoteKey, "narratio-session-*.yml")
|
||||
if err != nil {
|
||||
return nil, missingSessionConfigError(discoveredSession.Searched, fmt.Sprintf("remote session %q download failed: %v", remoteKey, err))
|
||||
}
|
||||
cleanup = func() error {
|
||||
if err := fileops.RemoveAllUnderRoot(filepath.Dir(sessionTempPath), sessionTempPath); err != nil {
|
||||
return fmt.Errorf("remove downloaded remote session config: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
sessionBytes, err := os.ReadFile(sessionTempPath)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read downloaded remote session %q: %w", sessionTempPath, err)
|
||||
return nil, fmt.Errorf("read downloaded remote session config: %w", err)
|
||||
}
|
||||
sessionCfg, err := config.LoadSessionBytesWithOptions("s3://"+s3BucketName(base.Pipeline)+"/"+remoteKey, sessionBytes, sessionOpts)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
return config.Resolve(
|
||||
cfg, err := config.Resolve(
|
||||
base.PipelinePath,
|
||||
base.Pipeline,
|
||||
base.CampaignPath,
|
||||
@@ -85,9 +136,14 @@ func loadCommandConfig(ctx context.Context, pipelineFlag, campaignFlag, campaign
|
||||
S3Key: remoteKey,
|
||||
S3Size: sessionInfo.Size,
|
||||
S3ETag: sessionInfo.ETag,
|
||||
SpoolPath: sessionTempPath,
|
||||
},
|
||||
)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
loaded = &commandConfig{Config: cfg, cleanup: cleanup}
|
||||
cleanup = nil
|
||||
return loaded, nil
|
||||
}
|
||||
|
||||
func loadPipelineCampaignConfig(pipelineFlag, campaignFlag, campaignFileFlag string) (*pipelineCampaignConfig, error) {
|
||||
|
||||
@@ -7,12 +7,14 @@ import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/notarius"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifactmodel"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifactpolicy"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||
@@ -25,6 +27,37 @@ type materializingNotariusRunner struct {
|
||||
failuresRemaining int
|
||||
}
|
||||
|
||||
type assertExtractionSourcesStage struct {
|
||||
keys []string
|
||||
runs *int
|
||||
}
|
||||
|
||||
func (s assertExtractionSourcesStage) Name() string { return "analyze" }
|
||||
|
||||
func (s assertExtractionSourcesStage) Run(_ context.Context, env *stage.Env, m *manifest.Manifest) (*stage.StageResult, error) {
|
||||
definitions := artifacts.ExtractionDefinitionsFromConfig(env.Config.Pipeline.Notarius)
|
||||
catalog, err := artifacts.BootstrapRuntimeCatalog(nil, env.EffectiveArtifacts, definitions)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
paths, err := env.ArtifactStore.EnsureLayoutFor(env.Config.Session.Campaign, env.Config.Session.SessionID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
catalog.HydrateExtractionArtifacts(paths, m, definitions)
|
||||
for _, key := range s.keys {
|
||||
sourceID := artifacts.ExtractionArtifactSourceID(key)
|
||||
entry, ok := catalog.Lookup(sourceID)
|
||||
if !ok || !entry.Available || entry.SourceID != sourceID || entry.Path == "" {
|
||||
return nil, fmt.Errorf("extraction source %q unavailable: %#v, present=%v", sourceID, entry, ok)
|
||||
}
|
||||
}
|
||||
if s.runs != nil {
|
||||
*s.runs = *s.runs + 1
|
||||
}
|
||||
return &stage.StageResult{}, nil
|
||||
}
|
||||
|
||||
func (r *materializingNotariusRunner) Run(_ context.Context, req notarius.RunRequest) (notarius.RunResult, error) {
|
||||
r.requests = append(r.requests, req)
|
||||
if r.failuresRemaining > 0 {
|
||||
@@ -38,32 +71,47 @@ func (r *materializingNotariusRunner) Run(_ context.Context, req notarius.RunReq
|
||||
return notarius.RunResult{}, err
|
||||
}
|
||||
for path, content := range map[string]string{
|
||||
filepath.Join(bundle, "index.json"): `{"manifest_file":"manifest.json"}`,
|
||||
filepath.Join(bundle, "manifest.json"): `{}`,
|
||||
filepath.Join(bundle, "rejected.json"): `{"rejected":[]}`,
|
||||
filepath.Join(bundle, "warnings.json"): `{"warnings":[]}`,
|
||||
filepath.Join(lanesDir, "npcs.json"): `{"npcs":[]}`,
|
||||
filepath.Join(bundle, "index.json"): `{"manifest_file":"manifest.json"}`,
|
||||
filepath.Join(bundle, "manifest.json"): `{}`,
|
||||
filepath.Join(bundle, "rejected.json"): `{"rejected":[]}`,
|
||||
filepath.Join(bundle, "warnings.json"): `{"schema_version":"notarius.warnings.v2","group_count":0,"occurrence_count":0,"groups":[]}`,
|
||||
filepath.Join(bundle, "diagnostics.json"): `{"schema_version":"notarius.diagnostics.v1","group_count":0,"occurrence_count":0,"truncated":false,"unrepresented_occurrence_count":0,"groups":[]}`,
|
||||
} {
|
||||
if err := os.WriteFile(path, []byte(content), 0o644); err != nil {
|
||||
return notarius.RunResult{}, err
|
||||
}
|
||||
}
|
||||
output := r.cfg.Outputs["npc_registry"]
|
||||
keys := make([]string, 0, len(r.cfg.Outputs))
|
||||
for key := range r.cfg.Outputs {
|
||||
keys = append(keys, key)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
lanes := make([]notarius.LaneDescriptor, 0, len(keys))
|
||||
for _, key := range keys {
|
||||
output := r.cfg.Outputs[key]
|
||||
filename := key + ".json"
|
||||
path := filepath.Join(lanesDir, filename)
|
||||
if err := os.WriteFile(path, []byte(`{"records":[]}`), 0o644); err != nil {
|
||||
return notarius.RunResult{}, err
|
||||
}
|
||||
lanes = append(lanes, notarius.LaneDescriptor{
|
||||
LaneID: output.LaneID, File: filepath.ToSlash(filepath.Join("lanes", filename)), Path: path,
|
||||
MediaType: output.MediaType, SchemaID: output.SchemaID,
|
||||
SchemaVersion: output.SchemaVersion, ModuleKey: output.ModuleKey,
|
||||
})
|
||||
}
|
||||
return notarius.RunResult{
|
||||
Receipt: notarius.Receipt{
|
||||
SchemaVersion: notarius.ReceiptSchemaVersion, RunID: externalRunID,
|
||||
PipelineID: req.PipelineID, OutputDirectory: bundle, IndexFile: "index.json",
|
||||
NormalizedOutputCount: 1, ValidationStatus: "valid",
|
||||
NormalizedOutputCount: len(lanes), ValidationStatus: "approved",
|
||||
},
|
||||
BundleRoot: bundle,
|
||||
Index: notarius.Index{
|
||||
Path: filepath.Join(bundle, "index.json"), RejectedPath: filepath.Join(bundle, "rejected.json"),
|
||||
WarningsPath: filepath.Join(bundle, "warnings.json"),
|
||||
Lanes: []notarius.LaneDescriptor{{
|
||||
LaneID: output.LaneID, File: "lanes/npcs.json", Path: filepath.Join(lanesDir, "npcs.json"),
|
||||
MediaType: output.MediaType, SchemaID: output.SchemaID,
|
||||
SchemaVersion: output.SchemaVersion, ModuleKey: output.ModuleKey,
|
||||
}},
|
||||
WarningsPath: filepath.Join(bundle, "warnings.json"),
|
||||
DiagnosticsPath: filepath.Join(bundle, "diagnostics.json"),
|
||||
Lanes: lanes,
|
||||
},
|
||||
}, nil
|
||||
}
|
||||
@@ -100,6 +148,185 @@ func TestExtractLifecycleDisabledThenEnabled(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractLifecyclePrepareBindsVerifiedReferenceSnapshots(t *testing.T) {
|
||||
cfg, env, runner := extractionLifecycleFixture(t, true)
|
||||
originalPaths := configureLifecycleReferences(t, cfg)
|
||||
|
||||
summary, err := executeStages(context.Background(), cfg, prepareExtractLifecyclePlan(t), RunOptions{Env: env})
|
||||
if err != nil {
|
||||
t.Fatalf("executeStages() error = %v", err)
|
||||
}
|
||||
if len(summary.Executed) != 2 || len(runner.requests) != 1 {
|
||||
t.Fatalf("summary = %#v requests=%d", summary, len(runner.requests))
|
||||
}
|
||||
paths, err := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root).EnsureLayoutFor(cfg.Session.Campaign, cfg.Session.SessionID)
|
||||
if err != nil {
|
||||
t.Fatalf("EnsureLayoutFor() error = %v", err)
|
||||
}
|
||||
want := []struct {
|
||||
selector string
|
||||
sourceID string
|
||||
filename string
|
||||
}{
|
||||
{selector: "glossary", sourceID: artifactpolicy.SourceInputGlossary, filename: "glossary.yml"},
|
||||
{selector: "party", sourceID: artifactpolicy.SourceInputParty, filename: "party.yml"},
|
||||
{selector: "players", sourceID: artifactpolicy.SourceInputPlayers, filename: "players.yml"},
|
||||
{selector: "spells", sourceID: artifactpolicy.SourceInputSpellCatalog, filename: "spell_catalog.json"},
|
||||
}
|
||||
request := runner.requests[0]
|
||||
if len(request.References) != len(want) {
|
||||
t.Fatalf("references = %#v", request.References)
|
||||
}
|
||||
loaded := loadLifecycleManifest(t, cfg)
|
||||
for index, expected := range want {
|
||||
binding := request.References[index]
|
||||
canonical := filepath.Join(paths.InputsDir, expected.filename)
|
||||
snapshot := filepath.Join(
|
||||
artifacts.SessionRunNotariusReferencesDirForCampaign(
|
||||
cfg.Pipeline.Workspace.Root, cfg.Session.Campaign, cfg.Session.SessionID, loaded.RunID,
|
||||
),
|
||||
expected.filename,
|
||||
)
|
||||
if binding.Selector != expected.selector || binding.Path != snapshot || binding.Path == canonical || binding.Path == originalPaths[expected.sourceID] {
|
||||
t.Fatalf("reference[%d] = %#v, want selector %q snapshot %q and not prepared/source paths", index, binding, expected.selector, snapshot)
|
||||
}
|
||||
}
|
||||
|
||||
extract := loaded.Stages["extract"]
|
||||
if extract == nil || extract.Status != manifest.StatusSucceeded || extract.Metadata["reference_count"] != float64(len(want)) {
|
||||
t.Fatalf("extract record = %#v", extract)
|
||||
}
|
||||
references, ok := extract.Metadata["references"].([]any)
|
||||
if !ok || len(references) != len(want) || len(references) > config.MaxNotariusReferenceBindings {
|
||||
t.Fatalf("reference metadata = %#v", extract.Metadata["references"])
|
||||
}
|
||||
for index, raw := range references {
|
||||
entry, ok := raw.(map[string]any)
|
||||
if !ok || len(entry) != 5 || entry["selector"] != want[index].selector || entry["source_id"] != want[index].sourceID {
|
||||
t.Fatalf("reference metadata[%d] = %#v", index, raw)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractLifecyclePreparedReferenceChangeRerunsExtractionAndInvalidatesDownstream(t *testing.T) {
|
||||
cfg, env, runner := extractionLifecycleFixture(t, true)
|
||||
originalPaths := configureLifecycleReferences(t, cfg)
|
||||
if _, err := executeStages(context.Background(), cfg, prepareExtractLifecyclePlan(t), RunOptions{Env: env}); err != nil {
|
||||
t.Fatalf("initial executeStages() error = %v", err)
|
||||
}
|
||||
before := loadLifecycleManifest(t, cfg)
|
||||
beforeChecksum := lifecycleInputChecksum(t, before, "party")
|
||||
for _, name := range []string{"render", "analyze", "publish"} {
|
||||
before.MarkStageSucceeded(name, time.Now().UTC(), nil)
|
||||
}
|
||||
if err := (&manifest.LocalStore{}).Save(context.Background(), manifestPathFor(cfg), before); err != nil {
|
||||
t.Fatalf("Save(downstream success) error = %v", err)
|
||||
}
|
||||
if err := os.WriteFile(originalPaths[artifactpolicy.SourceInputParty], []byte("changed party bytes\n"), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile(party source) error = %v", err)
|
||||
}
|
||||
|
||||
prepared := loadLifecycleManifest(t, cfg)
|
||||
prepare, err := stage.Select("prepare")
|
||||
if err != nil {
|
||||
t.Fatalf("stage.Select(prepare) error = %v", err)
|
||||
}
|
||||
if _, err := prepare.Run(context.Background(), env, prepared); err != nil {
|
||||
t.Fatalf("prepare.Run() error = %v", err)
|
||||
}
|
||||
if lifecycleInputChecksum(t, prepared, "party") == beforeChecksum {
|
||||
t.Fatal("prepared party checksum did not change")
|
||||
}
|
||||
if err := (&manifest.LocalStore{}).Save(context.Background(), manifestPathFor(cfg), prepared); err != nil {
|
||||
t.Fatalf("Save(reprepared manifest) error = %v", err)
|
||||
}
|
||||
|
||||
plan, err := BuildSingleStagePlan("extract")
|
||||
if err != nil {
|
||||
t.Fatalf("BuildSingleStagePlan(extract) error = %v", err)
|
||||
}
|
||||
run, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env})
|
||||
if err != nil {
|
||||
t.Fatalf("rerun executeStages() error = %v", err)
|
||||
}
|
||||
if len(run.Executed) != 1 || len(run.Skipped) != 0 || len(runner.requests) != 2 {
|
||||
t.Fatalf("rerun summary = %#v requests=%d", run, len(runner.requests))
|
||||
}
|
||||
after := loadLifecycleManifest(t, cfg)
|
||||
if after.Stages["extract"].Status != manifest.StatusSucceeded {
|
||||
t.Fatalf("extract status = %#v", after.Stages["extract"])
|
||||
}
|
||||
if after.Stages["render"] == nil || after.Stages["render"].Status != manifest.StatusSucceeded {
|
||||
t.Fatalf("render status = %#v, want succeeded sibling", after.Stages["render"])
|
||||
}
|
||||
for _, name := range []string{"analyze", "publish"} {
|
||||
if after.Stages[name] == nil || after.Stages[name].Status != manifest.StatusStale {
|
||||
t.Fatalf("%s status = %#v, want stale", name, after.Stages[name])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractLifecycleSessionOverrideBytesReachCanonicalReference(t *testing.T) {
|
||||
cfg, env, runner := extractionLifecycleFixture(t, true)
|
||||
configureLifecycleReferences(t, cfg)
|
||||
overridePath := filepath.Join(filepath.Dir(cfg.SessionPath), "session-party.yml")
|
||||
if err := os.WriteFile(overridePath, []byte("session override party\n"), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile(session override) error = %v", err)
|
||||
}
|
||||
cfg.StableInputs.PartyFile = config.ResolvedInputFile{
|
||||
Path: "./session-party.yml", ConfigPath: cfg.SessionPath, Source: "session_config",
|
||||
}
|
||||
cfg.Session.Inputs.PartyFile = "./session-party.yml"
|
||||
|
||||
if _, err := executeStages(context.Background(), cfg, prepareExtractLifecyclePlan(t), RunOptions{Env: env}); err != nil {
|
||||
t.Fatalf("executeStages() error = %v", err)
|
||||
}
|
||||
if len(runner.requests) != 1 {
|
||||
t.Fatalf("requests = %d", len(runner.requests))
|
||||
}
|
||||
var partyPath string
|
||||
for _, binding := range runner.requests[0].References {
|
||||
if binding.Selector == "party" {
|
||||
partyPath = binding.Path
|
||||
}
|
||||
}
|
||||
contents, err := os.ReadFile(partyPath)
|
||||
if err != nil {
|
||||
t.Fatalf("ReadFile(prepared party) error = %v", err)
|
||||
}
|
||||
if string(contents) != "session override party\n" || partyPath == overridePath {
|
||||
t.Fatalf("prepared party path=%q contents=%q override=%q", partyPath, contents, overridePath)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractLifecycleEmptyReferencesPreserveAllDndExtractionSources(t *testing.T) {
|
||||
cfg, env, runner := extractionLifecycleFixture(t, true)
|
||||
cfg.Pipeline.Notarius.Outputs = lifecycleDndOutputs()
|
||||
keys := make([]string, 0, len(cfg.Pipeline.Notarius.Outputs))
|
||||
for key := range cfg.Pipeline.Notarius.Outputs {
|
||||
keys = append(keys, key)
|
||||
}
|
||||
sort.Strings(keys)
|
||||
analyzeRuns := 0
|
||||
extractPlan, err := BuildSingleStagePlan("extract")
|
||||
if err != nil {
|
||||
t.Fatalf("BuildSingleStagePlan(extract) error = %v", err)
|
||||
}
|
||||
plan := append(extractPlan, assertExtractionSourcesStage{keys: keys, runs: &analyzeRuns})
|
||||
|
||||
summary, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env})
|
||||
if err != nil {
|
||||
t.Fatalf("executeStages() error = %v", err)
|
||||
}
|
||||
if len(summary.Executed) != 2 || len(runner.requests) != 1 || len(runner.requests[0].References) != 0 || analyzeRuns != 1 {
|
||||
t.Fatalf("summary=%#v requests=%#v analyze=%d", summary, runner.requests, analyzeRuns)
|
||||
}
|
||||
loaded := loadLifecycleManifest(t, cfg)
|
||||
if got := len(loaded.Stages["extract"].Outputs); got != len(keys)+1 {
|
||||
t.Fatalf("extract outputs = %d, want %d lanes plus index", got, len(keys))
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractLifecycleChangedOutcomeRerunsSucceededDownstream(t *testing.T) {
|
||||
cfg, env, runner := extractionLifecycleFixture(t, false)
|
||||
analyzeRuns := 0
|
||||
@@ -252,6 +479,32 @@ func TestExtractLifecycleSkipsCurrentResumableResult(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractLifecycleRerunsAfterDirectTranscriptChange(t *testing.T) {
|
||||
cfg, env, runner := extractionLifecycleFixture(t, true)
|
||||
analyzeRuns := 0
|
||||
plan := extractionLifecyclePlan(t, &analyzeRuns)
|
||||
if _, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env}); err != nil {
|
||||
t.Fatalf("first executeStages() error = %v", err)
|
||||
}
|
||||
|
||||
persisted := loadLifecycleManifest(t, cfg)
|
||||
trimmed := persisted.Stages["trim"].Outputs[0].LocalPath
|
||||
if err := os.WriteFile(trimmed, []byte(`{"segments":[{"id":"changed"}]}`), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile(trimmed transcript) error = %v", err)
|
||||
}
|
||||
if err := (&manifest.LocalStore{}).Save(context.Background(), manifestPathFor(cfg), persisted); err != nil {
|
||||
t.Fatalf("Save(mutated) error = %v", err)
|
||||
}
|
||||
|
||||
rerun, err := executeStages(context.Background(), cfg, plan, RunOptions{Env: env})
|
||||
if err != nil {
|
||||
t.Fatalf("rerun executeStages() error = %v", err)
|
||||
}
|
||||
if len(rerun.Executed) != 2 || len(rerun.Skipped) != 0 || len(runner.requests) != 2 || analyzeRuns != 2 {
|
||||
t.Fatalf("rerun summary = %#v requests=%d analyze=%d", rerun, len(runner.requests), analyzeRuns)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractLifecycleResumesAndRerunsObsoleteResults(t *testing.T) {
|
||||
for _, test := range []struct {
|
||||
name string
|
||||
@@ -368,6 +621,73 @@ func markLifecycleStageSucceeded(t *testing.T, cfg *config.Config, name string)
|
||||
}
|
||||
}
|
||||
|
||||
func prepareExtractLifecyclePlan(t *testing.T) []stage.Stage {
|
||||
t.Helper()
|
||||
prepare, err := BuildSingleStagePlan("prepare")
|
||||
if err != nil {
|
||||
t.Fatalf("BuildSingleStagePlan(prepare) error = %v", err)
|
||||
}
|
||||
extract, err := BuildSingleStagePlan("extract")
|
||||
if err != nil {
|
||||
t.Fatalf("BuildSingleStagePlan(extract) error = %v", err)
|
||||
}
|
||||
return append(prepare, extract...)
|
||||
}
|
||||
|
||||
func configureLifecycleReferences(t *testing.T, cfg *config.Config) map[string]string {
|
||||
t.Helper()
|
||||
cfg.Pipeline.Notarius.References = map[string]string{
|
||||
"party": artifactpolicy.SourceInputParty,
|
||||
"players": artifactpolicy.SourceInputPlayers,
|
||||
"glossary": artifactpolicy.SourceInputGlossary,
|
||||
"spells": artifactpolicy.SourceInputSpellCatalog,
|
||||
}
|
||||
spellPath := filepath.Join(filepath.Dir(cfg.CampaignPath), "spells.json")
|
||||
if err := os.WriteFile(spellPath, []byte(`{"spells":[]}`+"\n"), 0o644); err != nil {
|
||||
t.Fatalf("WriteFile(spell catalog) error = %v", err)
|
||||
}
|
||||
cfg.StableInputs.SpellCatalogFile = config.ResolvedInputFile{
|
||||
Path: "./spells.json", ConfigPath: cfg.CampaignPath, Source: "campaign_config",
|
||||
}
|
||||
cfg.Session.Inputs.SpellCatalogFile = "./spells.json"
|
||||
|
||||
return map[string]string{
|
||||
artifactpolicy.SourceInputParty: filepath.Join(filepath.Dir(cfg.CampaignPath), "party.yml"),
|
||||
artifactpolicy.SourceInputPlayers: filepath.Join(filepath.Dir(cfg.CampaignPath), "players.yml"),
|
||||
artifactpolicy.SourceInputGlossary: filepath.Join(filepath.Dir(cfg.CampaignPath), "glossary.yml"),
|
||||
artifactpolicy.SourceInputSpellCatalog: spellPath,
|
||||
}
|
||||
}
|
||||
|
||||
func lifecycleInputChecksum(t *testing.T, m *manifest.Manifest, kind string) string {
|
||||
t.Helper()
|
||||
for _, input := range m.Inputs {
|
||||
if input.Kind == kind {
|
||||
if strings.TrimSpace(input.Checksum) == "" {
|
||||
t.Fatalf("input %q has no checksum: %#v", kind, input)
|
||||
}
|
||||
return input.Checksum
|
||||
}
|
||||
}
|
||||
t.Fatalf("manifest input %q not found: %#v", kind, m.Inputs)
|
||||
return ""
|
||||
}
|
||||
|
||||
func lifecycleDndOutputs() map[string]config.NotariusOutputConfig {
|
||||
return map[string]config.NotariusOutputConfig{
|
||||
"item_registry": {LaneID: "item-registry", MediaType: "application/json", SchemaID: "notarius.dnd.item_registry", SchemaVersion: "v1", ModuleKey: "dnd/item-registry"},
|
||||
"npc_registry": {LaneID: "npc-registry", MediaType: "application/json", SchemaID: "notarius.dnd.npc_registry", SchemaVersion: "v1", ModuleKey: "dnd/npc-registry"},
|
||||
"location_registry": {LaneID: "location-registry", MediaType: "application/json", SchemaID: "notarius.dnd.location_registry", SchemaVersion: "v1", ModuleKey: "dnd/location-registry"},
|
||||
"scene_descriptions": {LaneID: "scene-descriptions", MediaType: "application/json", SchemaID: "notarius.dnd.scene_descriptions", SchemaVersion: "v1", ModuleKey: "dnd/scene-descriptions"},
|
||||
"item_occurrences": {LaneID: "item-occurrences", MediaType: "application/json", SchemaID: "notarius.dnd.item_occurrences", SchemaVersion: "v1", ModuleKey: "dnd/item-occurrences"},
|
||||
"spells": {LaneID: "spells", MediaType: "application/json", SchemaID: "notarius.dnd.spells", SchemaVersion: "v1", ModuleKey: "dnd/spells"},
|
||||
"combat_turns": {LaneID: "combat-turns", MediaType: "application/json", SchemaID: "notarius.dnd.combat_turns", SchemaVersion: "v1", ModuleKey: "dnd/combat-turns"},
|
||||
"npc_occurrences": {LaneID: "npc-occurrences", MediaType: "application/json", SchemaID: "notarius.dnd.npc_occurrences", SchemaVersion: "v1", ModuleKey: "dnd/npc-occurrences"},
|
||||
"location_occurrences": {LaneID: "location-occurrences", MediaType: "application/json", SchemaID: "notarius.dnd.location_occurrences", SchemaVersion: "v1", ModuleKey: "dnd/location-occurrences"},
|
||||
"enemy_events": {LaneID: "enemy-events", MediaType: "application/json", SchemaID: "notarius.dnd.enemy_events", SchemaVersion: "v1", ModuleKey: "dnd/enemy-events"},
|
||||
}
|
||||
}
|
||||
|
||||
func extractionLifecycleFixture(t *testing.T, enabled bool) (*config.Config, *stage.Env, *materializingNotariusRunner) {
|
||||
t.Helper()
|
||||
cfg := testConfig(t)
|
||||
|
||||
@@ -142,24 +142,3 @@ func commandObjectStoreTestConfig(secretsDir string) *config.Config {
|
||||
}
|
||||
return cfg
|
||||
}
|
||||
|
||||
func restoreEnvAfterTest(t *testing.T, names ...string) {
|
||||
t.Helper()
|
||||
originals := make(map[string]string, len(names))
|
||||
present := make(map[string]bool, len(names))
|
||||
for _, name := range names {
|
||||
value, ok := os.LookupEnv(name)
|
||||
originals[name] = value
|
||||
present[name] = ok
|
||||
_ = os.Unsetenv(name)
|
||||
}
|
||||
t.Cleanup(func() {
|
||||
for _, name := range names {
|
||||
if present[name] {
|
||||
_ = os.Setenv(name, originals[name])
|
||||
} else {
|
||||
_ = os.Unsetenv(name)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
@@ -14,27 +14,24 @@ import (
|
||||
)
|
||||
|
||||
func buildHelperArtifactCatalog(cfg *config.Config, m *manifest.Manifest) (*artifacts.ArtifactCatalog, error) {
|
||||
catalog := artifacts.NewArtifactCatalog()
|
||||
if err := catalog.RegisterBuiltIns(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
configured := map[string]artifacts.ConfiguredArtifactDefinition{}
|
||||
configured := artifacts.ConfiguredArtifactDefinitions(nil)
|
||||
if cfg.Pipeline.Scriptorium != nil {
|
||||
for key, item := range cfg.Pipeline.Scriptorium.Artifacts {
|
||||
configured[key] = artifacts.ConfiguredArtifactDefinition{Enabled: item.Enabled, OutputPath: item.OutputPath}
|
||||
}
|
||||
}
|
||||
if err := catalog.RegisterConfiguredArtifacts(configured, nil); err != nil {
|
||||
return nil, err
|
||||
configured = artifacts.ConfiguredArtifactDefinitions(cfg.Pipeline.Scriptorium.Artifacts)
|
||||
}
|
||||
extractionDefinitions := artifacts.ExtractionDefinitionsFromConfig(cfg.Pipeline.Notarius)
|
||||
if err := catalog.RegisterExtractionArtifacts(extractionDefinitions); err != nil {
|
||||
effective, err := artifacts.ResolveEffectiveArtifactSet(configured, nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
catalog, err := artifacts.BootstrapRuntimeCatalog(configured, effective, extractionDefinitions)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
paths := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root).SessionPathsFor(cfg.Session.Campaign, cfg.Session.SessionID)
|
||||
if cfg.Pipeline.Notarius != nil && cfg.Pipeline.Notarius.Enabled {
|
||||
catalog.HydrateExtractionArtifacts(paths, m, extractionDefinitions)
|
||||
}
|
||||
catalog.HydrateAnalyzeArtifacts(paths, m, configured)
|
||||
return catalog, nil
|
||||
}
|
||||
|
||||
@@ -47,7 +44,11 @@ func writeArtifactList(out io.Writer, cfg *config.Config, catalog *artifacts.Art
|
||||
writeArtifactLine(out, artifacts.ArtifactBoundsSession, lockSet)
|
||||
fmt.Fprintln(out, "Configured:")
|
||||
for _, entry := range catalog.ListConfigured() {
|
||||
writeArtifactLine(out, entry.SourceID, lockSet)
|
||||
state := "unavailable"
|
||||
if entry.Available {
|
||||
state = "available"
|
||||
}
|
||||
writeExtractionArtifactLine(out, entry.SourceID, state, entry.Provenance, lockSet)
|
||||
}
|
||||
fmt.Fprintln(out, "Extraction:")
|
||||
for _, entry := range catalog.ListExtraction() {
|
||||
@@ -58,8 +59,10 @@ func writeArtifactList(out io.Writer, cfg *config.Config, catalog *artifacts.Art
|
||||
writeExtractionArtifactLine(out, entry.SourceID, state, entry.Provenance, lockSet)
|
||||
}
|
||||
fmt.Fprintln(out, "Previous-session:")
|
||||
for _, req := range artifacts.CollectPreviousArtifactRequirements(configuredScriptoriumArtifacts(cfg)) {
|
||||
fmt.Fprintf(out, "- %s required=%t\n", artifactpolicy.PreviousSessionSourceID(req.Name), req.Required)
|
||||
if effective, err := resolveEffectiveArtifacts(cfg, nil); err == nil {
|
||||
for _, req := range artifacts.CollectPreviousArtifactRequirements(configuredScriptoriumArtifacts(cfg), effective) {
|
||||
fmt.Fprintf(out, "- %s required=%t\n", artifactpolicy.PreviousSessionSourceID(req.Name), req.Required)
|
||||
}
|
||||
}
|
||||
fmt.Fprintln(out, "Published:")
|
||||
for _, rule := range cfg.Pipeline.Publish.Outputs {
|
||||
@@ -129,6 +132,47 @@ func remotePublishedOutputAvailability(ctx context.Context, cfg *config.Config,
|
||||
return out
|
||||
}
|
||||
|
||||
func remotePublishedOutputAvailabilityForCurrent(
|
||||
ctx context.Context,
|
||||
cfg *config.Config,
|
||||
store storage.ObjectStore,
|
||||
catalog *artifacts.ArtifactCatalog,
|
||||
current *RemoteCurrentState,
|
||||
) map[string]string {
|
||||
if current == nil || current.Commit == nil {
|
||||
return remotePublishedOutputAvailability(ctx, cfg, store, catalog)
|
||||
}
|
||||
out := map[string]string{}
|
||||
runPrefix := artifacts.S3RunPrefix(current.SessionPrefix, current.RunID)
|
||||
for _, rule := range cfg.Pipeline.Publish.Outputs {
|
||||
source := strings.TrimSpace(rule.Source)
|
||||
dest, _, err := helperPublishedOutputDest(rule, catalog)
|
||||
if err != nil {
|
||||
out[publishedOutputRemoteStateKey(source, "")] = "remote=error"
|
||||
continue
|
||||
}
|
||||
key := artifacts.S3RunRelativeDestinationKey(runPrefix, dest)
|
||||
if committedPublishedOutput(current.Commit, key) {
|
||||
out[publishedOutputRemoteStateKey(source, dest)] = "remote=published"
|
||||
} else {
|
||||
out[publishedOutputRemoteStateKey(source, dest)] = "remote=missing"
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func committedPublishedOutput(commit *artifacts.RemoteCommitManifest, key string) bool {
|
||||
if commit == nil {
|
||||
return false
|
||||
}
|
||||
for _, artifact := range commit.Artifacts {
|
||||
if artifact.Type == artifacts.RemoteArtifactTypePublishedOutput && artifact.DestinationKey == key {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func helperPublishedOutputDest(rule config.PublishOutputRule, catalog *artifacts.ArtifactCatalog) (string, bool, error) {
|
||||
source := strings.TrimSpace(rule.Source)
|
||||
normalized, err := artifactpolicy.ResolvePublishedDestinationWithExtractions(
|
||||
|
||||
54
internal/app/operator_artifact_rendering_test.go
Normal file
54
internal/app/operator_artifact_rendering_test.go
Normal file
@@ -0,0 +1,54 @@
|
||||
package app
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/manifest"
|
||||
)
|
||||
|
||||
func TestBuildHelperArtifactCatalogUsesAnalyzeManifestEvidence(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
cfg := &config.Config{
|
||||
Pipeline: &config.PipelineConfig{
|
||||
Workspace: config.WorkspaceConfig{Root: root},
|
||||
Scriptorium: &config.ScriptoriumConfig{Artifacts: map[string]config.ScriptoriumArtifactConfig{
|
||||
"session_recap": {Enabled: true, OutputPath: "artifacts/session_recap.md"},
|
||||
}},
|
||||
},
|
||||
Session: &config.SessionConfig{Campaign: "campaign", SessionID: "session"},
|
||||
}
|
||||
paths := artifacts.NewLocalStore(root).SessionPathsFor("campaign", "session")
|
||||
outputPath := filepath.Join(paths.ArtifactsDir, "session_recap.md")
|
||||
if err := os.MkdirAll(filepath.Dir(outputPath), 0o755); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
body := []byte("# recap\n")
|
||||
if err := os.WriteFile(outputPath, body, 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m := manifest.New("session", time.Date(2026, 8, 29, 12, 0, 0, 0, time.UTC))
|
||||
|
||||
incidental, err := buildHelperArtifactCatalog(cfg, m)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
entry, _ := incidental.Lookup(artifacts.ConfiguredArtifactSourceID("session_recap"))
|
||||
if entry.Available {
|
||||
t.Fatal("operator catalog advertised incidental configured artifact")
|
||||
}
|
||||
|
||||
setAppAnalyzeEvidence(m, "session_recap", "artifacts/session_recap.md", body)
|
||||
current, err := buildHelperArtifactCatalog(cfg, m)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
entry, _ = current.Lookup(artifacts.ConfiguredArtifactSourceID("session_recap"))
|
||||
if !entry.Available || entry.Provenance != artifacts.ArtifactProvenanceCurrentAnalyzeManifest {
|
||||
t.Fatalf("operator catalog entry = %#v", entry)
|
||||
}
|
||||
}
|
||||
@@ -22,10 +22,11 @@ func ArtifactsList(ctx context.Context, args []string, out io.Writer) error {
|
||||
if strings.TrimSpace(flags.sessionID) == "" {
|
||||
return fmt.Errorf("artifacts list: session_id is required")
|
||||
}
|
||||
cfg, store, locks, m, err := loadHelperContext(ctx, flags, remote)
|
||||
cfg, store, locks, m, cleanup, err := loadHelperContext(ctx, flags, remote)
|
||||
if err != nil {
|
||||
return fmt.Errorf("artifacts list: %w", err)
|
||||
}
|
||||
defer cleanup()
|
||||
catalog, err := buildHelperArtifactCatalog(cfg, m)
|
||||
if err != nil {
|
||||
return fmt.Errorf("artifacts list: %w", err)
|
||||
|
||||
@@ -90,33 +90,41 @@ func Artifacts(ctx context.Context, args []string, out io.Writer) error {
|
||||
}
|
||||
}
|
||||
|
||||
func loadHelperContext(ctx context.Context, flags commonConfigFlags, needStore bool) (*config.Config, storage.ObjectStore, *effectiveLocks, *manifest.Manifest, error) {
|
||||
cfg, err := loadCommandConfig(ctx, flags.pipelinePath, flags.campaignPath, flags.campaignFilePath, flags.sessionPath, flags.sessionOptions())
|
||||
func loadHelperContext(ctx context.Context, flags commonConfigFlags, needStore bool) (*config.Config, storage.ObjectStore, *effectiveLocks, *manifest.Manifest, func(), error) {
|
||||
loaded, err := loadCommandConfig(ctx, flags.pipelinePath, flags.campaignPath, flags.campaignFilePath, flags.sessionPath, flags.sessionOptions())
|
||||
if err != nil {
|
||||
return nil, nil, nil, nil, err
|
||||
return nil, nil, nil, nil, nil, err
|
||||
}
|
||||
release := true
|
||||
defer func() {
|
||||
if release {
|
||||
_ = loaded.Close()
|
||||
}
|
||||
}()
|
||||
cfg := loaded.Config
|
||||
if err := config.Validate(cfg); err != nil {
|
||||
return nil, nil, nil, nil, err
|
||||
return nil, nil, nil, nil, nil, err
|
||||
}
|
||||
var store storage.ObjectStore
|
||||
if needStore {
|
||||
store, err = newCommandObjectStore(ctx, cfg, nil)
|
||||
if err != nil {
|
||||
return nil, nil, nil, nil, err
|
||||
return nil, nil, nil, nil, nil, err
|
||||
}
|
||||
} else {
|
||||
store, _ = objectStoreIfConfigured(ctx, cfg)
|
||||
}
|
||||
locks, err := loadEffectiveLocks(ctx, cfg, store)
|
||||
if err != nil {
|
||||
return nil, nil, nil, nil, err
|
||||
return nil, nil, nil, nil, nil, err
|
||||
}
|
||||
paths := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root).SessionPathsFor(cfg.Session.Campaign, cfg.Session.SessionID)
|
||||
m, err := loadLocalManifest(ctx, paths.ManifestPath)
|
||||
if err != nil {
|
||||
return nil, nil, nil, nil, err
|
||||
return nil, nil, nil, nil, nil, err
|
||||
}
|
||||
return cfg, store, locks, m, nil
|
||||
release = false
|
||||
return cfg, store, locks, m, func() { _ = loaded.Close() }, nil
|
||||
}
|
||||
|
||||
func objectStoreIfConfigured(ctx context.Context, cfg *config.Config) (storage.ObjectStore, error) {
|
||||
|
||||
@@ -3,10 +3,12 @@ package app
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -189,8 +191,8 @@ func TestExecuteSessionInitRemoteLoadsSecretsBeforeObjectStoreInit(t *testing.T)
|
||||
secretKeyEnv := "NARRATIO_TEST_SESSION_INIT_OBJECT_SECRET"
|
||||
restoreEnvAfterTest(t, accessKeyEnv, secretKeyEnv)
|
||||
secretsDir := t.TempDir()
|
||||
mustWriteTestFile(t, filepath.Join(secretsDir, accessKeyEnv), "test-key-id\n")
|
||||
mustWriteTestFile(t, filepath.Join(secretsDir, secretKeyEnv), "test-secret\n")
|
||||
mustWriteSecretFile(t, filepath.Join(secretsDir, accessKeyEnv), "test-key-id\n")
|
||||
mustWriteSecretFile(t, filepath.Join(secretsDir, secretKeyEnv), "test-secret\n")
|
||||
addSecretsToPipelineConfig(t, pipelinePath, secretsDir, accessKeyEnv, secretKeyEnv)
|
||||
|
||||
fake := &storage.FakeBackend{}
|
||||
@@ -417,8 +419,8 @@ func TestExecuteSessionValidateLoadsSecretsBeforeObjectStoreInit(t *testing.T) {
|
||||
secretKeyEnv := "NARRATIO_TEST_VALIDATE_OBJECT_SECRET"
|
||||
restoreEnvAfterTest(t, accessKeyEnv, secretKeyEnv)
|
||||
secretsDir := t.TempDir()
|
||||
mustWriteTestFile(t, filepath.Join(secretsDir, accessKeyEnv), "test-key-id\n")
|
||||
mustWriteTestFile(t, filepath.Join(secretsDir, secretKeyEnv), "test-secret\n")
|
||||
mustWriteSecretFile(t, filepath.Join(secretsDir, accessKeyEnv), "test-key-id\n")
|
||||
mustWriteSecretFile(t, filepath.Join(secretsDir, secretKeyEnv), "test-secret\n")
|
||||
addSecretsToPipelineConfig(t, pipelinePath, secretsDir, accessKeyEnv, secretKeyEnv)
|
||||
if err := os.WriteFile(sessionPath, []byte(`session_id: 2026-05-03
|
||||
inputs:
|
||||
@@ -456,6 +458,63 @@ inputs:
|
||||
}
|
||||
}
|
||||
|
||||
func TestOperatorCommandsReportConfiguredSpellCatalog(t *testing.T) {
|
||||
workspaceRoot := t.TempDir()
|
||||
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
|
||||
campaignBytes, err := os.ReadFile(campaignPath)
|
||||
if err != nil {
|
||||
t.Fatalf("read campaign: %v", err)
|
||||
}
|
||||
campaignYAML := strings.Replace(string(campaignBytes), " party_file: ./party.yml\n", " party_file: ./party.yml\n spell_catalog_file: ./spells.json\n", 1)
|
||||
if err := os.WriteFile(campaignPath, []byte(campaignYAML), 0o644); err != nil {
|
||||
t.Fatalf("write campaign: %v", err)
|
||||
}
|
||||
spellPath := filepath.Join(filepath.Dir(campaignPath), "spells.json")
|
||||
mustWriteTestFile(t, spellPath, "{\"spells\":[]}\n")
|
||||
|
||||
fake := &storage.FakeBackend{}
|
||||
var storeInitCalls int
|
||||
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{sessionPath})
|
||||
commonArgs := []string{
|
||||
"2026-05-03",
|
||||
"--config", pipelinePath,
|
||||
"--campaign-file", campaignPath,
|
||||
"--session", sessionPath,
|
||||
}
|
||||
|
||||
var stdout bytes.Buffer
|
||||
var stderr bytes.Buffer
|
||||
validateArgs := append([]string{"session", "validate"}, commonArgs...)
|
||||
if code := Execute(validateArgs, &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("validate exit code = %d, want 0; stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
if !strings.Contains(stdout.String(), "OK inputs spell_catalog: "+spellPath) {
|
||||
t.Fatalf("validate stdout = %q, want spell catalog finding", stdout.String())
|
||||
}
|
||||
|
||||
stdout.Reset()
|
||||
stderr.Reset()
|
||||
statusArgs := append([]string{"session", "status"}, commonArgs...)
|
||||
if code := Execute(statusArgs, &stdout, &stderr); code != 0 {
|
||||
t.Fatalf("status exit code = %d, want 0; stdout=%q stderr=%q", code, stdout.String(), stderr.String())
|
||||
}
|
||||
if !strings.Contains(stdout.String(), "Stable input spell_catalog: "+spellPath) {
|
||||
t.Fatalf("status stdout = %q, want spell catalog inventory", stdout.String())
|
||||
}
|
||||
|
||||
if err := os.Remove(spellPath); err != nil {
|
||||
t.Fatalf("remove spell catalog: %v", err)
|
||||
}
|
||||
stdout.Reset()
|
||||
stderr.Reset()
|
||||
if code := Execute(validateArgs, &stdout, &stderr); code == 0 {
|
||||
t.Fatalf("validate missing spell catalog exit code = 0; stdout=%q", stdout.String())
|
||||
}
|
||||
if !strings.Contains(stdout.String(), "ERROR inputs spell_catalog missing:") {
|
||||
t.Fatalf("validate stdout = %q, want missing spell catalog finding", stdout.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteLocksAddListAndRemoveUseRemoteLockStore(t *testing.T) {
|
||||
workspaceRoot := t.TempDir()
|
||||
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
|
||||
@@ -522,6 +581,123 @@ func TestExecuteLocksAddListAndRemoveUseRemoteLockStore(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestMutateRemoteLockStoreRetainsConcurrentUpdates(t *testing.T) {
|
||||
cfg := &config.Config{
|
||||
Pipeline: &config.PipelineConfig{Storage: config.StorageConfig{S3: &config.StorageS3Config{Bucket: "bucket", RootPrefix: "root"}}},
|
||||
Session: &config.SessionConfig{Campaign: "campaign", SessionID: "session"},
|
||||
}
|
||||
fake := &storage.FakeBackend{}
|
||||
arrived := make(chan struct{}, 2)
|
||||
release := make(chan struct{})
|
||||
var hookMu sync.Mutex
|
||||
hookCalls := 0
|
||||
fake.UploadHook = func(storage.FakeUploadCall) error {
|
||||
hookMu.Lock()
|
||||
hookCalls++
|
||||
call := hookCalls
|
||||
hookMu.Unlock()
|
||||
if call <= 2 {
|
||||
arrived <- struct{}{}
|
||||
<-release
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
mutate := func(source string) error {
|
||||
return mutateRemoteLockStore(context.Background(), cfg, fake, func(lockStore *config.PublishLockStore) error {
|
||||
set := lockSourceSet(lockStore.Locks)
|
||||
set[source] = config.PublishLockRule{Source: source}
|
||||
lockStore.Locks = lockMapValues(set)
|
||||
return nil
|
||||
})
|
||||
}
|
||||
errs := make(chan error, 2)
|
||||
go func() { errs <- mutate("narratio.transcript.final") }()
|
||||
go func() { errs <- mutate("narratio.transcript.final_trimmed") }()
|
||||
<-arrived
|
||||
<-arrived
|
||||
close(release)
|
||||
if err := <-errs; err != nil {
|
||||
t.Fatalf("first concurrent mutation error = %v", err)
|
||||
}
|
||||
if err := <-errs; err != nil {
|
||||
t.Fatalf("second concurrent mutation error = %v", err)
|
||||
}
|
||||
|
||||
locks, _, _, err := loadRemoteLockStore(context.Background(), cfg, fake)
|
||||
if err != nil {
|
||||
t.Fatalf("loadRemoteLockStore() error = %v", err)
|
||||
}
|
||||
if len(locks.Locks) != 2 || locks.Locks[0].Source != "narratio.transcript.final" || locks.Locks[1].Source != "narratio.transcript.final_trimmed" {
|
||||
t.Fatalf("remote locks = %#v, want both concurrent updates", locks.Locks)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLoadRemoteLockStoreAcceptsExactLimitAndRejectsLimitPlusOne(t *testing.T) {
|
||||
cfg := &config.Config{
|
||||
Pipeline: &config.PipelineConfig{Storage: config.StorageConfig{S3: &config.StorageS3Config{Bucket: "bucket", RootPrefix: "root"}}},
|
||||
Session: &config.SessionConfig{Campaign: "campaign", SessionID: "session"},
|
||||
}
|
||||
key, err := remoteLocksKey(cfg)
|
||||
if err != nil {
|
||||
t.Fatalf("remoteLocksKey() error = %v", err)
|
||||
}
|
||||
encoded, err := config.MarshalPublishLockStore(&config.PublishLockStore{})
|
||||
if err != nil {
|
||||
t.Fatalf("MarshalPublishLockStore() error = %v", err)
|
||||
}
|
||||
exact := append(append([]byte(nil), encoded...), bytes.Repeat([]byte(" "), int(MaxRemoteLockStoreBytes)-len(encoded))...)
|
||||
store := &storage.FakeBackend{}
|
||||
store.SeedObject(storage.FakeObject{Key: key, Data: exact, ETag: "lock-generation"})
|
||||
|
||||
locks, gotKey, generation, err := loadRemoteLockStore(context.Background(), cfg, store)
|
||||
if err != nil {
|
||||
t.Fatalf("loadRemoteLockStore() exact-limit error = %v", err)
|
||||
}
|
||||
if locks == nil || gotKey != key || generation != "lock-generation" {
|
||||
t.Fatalf("loadRemoteLockStore() = (%#v, %q, %q), want decoded locks and opened generation", locks, gotKey, generation)
|
||||
}
|
||||
if len(store.Downloads) != 0 || len(store.Reads) != 1 || store.Reads[0].Key != key {
|
||||
t.Fatalf("lock transfers reads=%#v downloads=%#v, want one direct read", store.Reads, store.Downloads)
|
||||
}
|
||||
|
||||
store.SeedObject(storage.FakeObject{Key: key, Data: append(exact, ' '), ETag: "new-generation"})
|
||||
_, _, _, err = loadRemoteLockStore(context.Background(), cfg, store)
|
||||
if err == nil || !strings.Contains(err.Error(), "remote lock control object") || !strings.Contains(err.Error(), key) || !strings.Contains(err.Error(), fmt.Sprint(MaxRemoteLockStoreBytes)) {
|
||||
t.Fatalf("loadRemoteLockStore() limit-plus-one error = %v, want category, key, and limit", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLoadRemoteLockStoreRejectsMalformedYAML(t *testing.T) {
|
||||
cfg := &config.Config{
|
||||
Pipeline: &config.PipelineConfig{Storage: config.StorageConfig{S3: &config.StorageS3Config{Bucket: "bucket", RootPrefix: "root"}}},
|
||||
Session: &config.SessionConfig{Campaign: "campaign", SessionID: "session"},
|
||||
}
|
||||
key, err := remoteLocksKey(cfg)
|
||||
if err != nil {
|
||||
t.Fatalf("remoteLocksKey() error = %v", err)
|
||||
}
|
||||
store := &storage.FakeBackend{}
|
||||
store.SeedObject(storage.FakeObject{Key: key, Data: []byte("locks: [\n"), ETag: "lock-generation"})
|
||||
|
||||
if _, _, _, err := loadRemoteLockStore(context.Background(), cfg, store); err == nil {
|
||||
t.Fatal("loadRemoteLockStore() error = nil, want malformed YAML failure")
|
||||
}
|
||||
}
|
||||
|
||||
func TestMutateRemoteLockStoreHonorsCancellation(t *testing.T) {
|
||||
cfg := &config.Config{
|
||||
Pipeline: &config.PipelineConfig{Storage: config.StorageConfig{S3: &config.StorageS3Config{Bucket: "bucket", RootPrefix: "root"}}},
|
||||
Session: &config.SessionConfig{Campaign: "campaign", SessionID: "session"},
|
||||
}
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
err := mutateRemoteLockStore(ctx, cfg, &storage.FakeBackend{}, func(*config.PublishLockStore) error { return nil })
|
||||
if !errors.Is(err, context.Canceled) {
|
||||
t.Fatalf("mutateRemoteLockStore() error = %v, want context cancellation", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteLocksAddDuplicateRequiresForce(t *testing.T) {
|
||||
workspaceRoot := t.TempDir()
|
||||
pipelinePath, campaignPath, sessionPath := writeValidConfigFiles(t, workspaceRoot)
|
||||
@@ -979,9 +1155,6 @@ func TestExecuteStatusReportsRemoteArtifactCatalogErrorsWithoutFailing(t *testin
|
||||
if !strings.Contains(out, "Remote outputs:") || !strings.Contains(out, "narratio.transcript.final_trimmed remote=error") {
|
||||
t.Fatalf("stdout = %q, want remote output error state", out)
|
||||
}
|
||||
if !strings.Contains(out, "Publish locks: error:") {
|
||||
t.Fatalf("stdout = %q, want publish locks error", out)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteStatusReportsMissingRemoteCurrentStateWithoutFailing(t *testing.T) {
|
||||
@@ -1032,6 +1205,36 @@ func TestExecuteStatusReportsPreviousStateReadinessWithoutFailing(t *testing.T)
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteStatusDetectsMissingRequiredPreviousArtifact(t *testing.T) {
|
||||
workspaceRoot := t.TempDir()
|
||||
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
|
||||
replaceInFileOrFatal(t, pipelinePath, "source: narratio.artifact.session_recap", "source: narratio.previous_session.artifact.session_recap")
|
||||
replaceInFileOrFatal(t, sessionPath, "session_id: 2026-05-03\n", "session_id: 2026-05-03\nprevious_session_id: 2026-04-26\n")
|
||||
cfg, err := config.LoadWithSessionOptions(pipelinePath, campaignPath, sessionPath, config.SessionLoadOptions{})
|
||||
if err != nil {
|
||||
t.Fatalf("LoadWithSessionOptions() error = %v", err)
|
||||
}
|
||||
fake := &storage.FakeBackend{}
|
||||
seedRestorePreviousCurrentManifestOnly(t, fake, cfg)
|
||||
var storeInitCalls int
|
||||
restoreAppConfigTestGlobals(t, fake, &storeInitCalls, []string{sessionPath})
|
||||
|
||||
var stdout bytes.Buffer
|
||||
var stderr bytes.Buffer
|
||||
code := Execute([]string{
|
||||
"session", "status", "2026-05-03",
|
||||
"--config", pipelinePath,
|
||||
"--campaign-file", campaignPath,
|
||||
"--session", sessionPath,
|
||||
}, &stdout, &stderr)
|
||||
if code != 0 {
|
||||
t.Fatalf("exit code = %d, want 0; stderr=%q", code, stderr.String())
|
||||
}
|
||||
if !strings.Contains(stdout.String(), `Previous-session artifacts: unavailable: remote required previous-session artifact "session_recap" object missing`) {
|
||||
t.Fatalf("stdout = %q, want missing required previous artifact", stdout.String())
|
||||
}
|
||||
}
|
||||
|
||||
func TestExecuteSessionValidateReportsPreviousStateFindingAndReturnsFindingError(t *testing.T) {
|
||||
workspaceRoot := t.TempDir()
|
||||
pipelinePath, campaignPath, sessionPath := writeValidConfigFilesWithScriptoriumArtifacts(t, workspaceRoot)
|
||||
@@ -1140,8 +1343,10 @@ func writeOperatorExtractionManifest(t *testing.T, workspaceRoot string) string
|
||||
bundleRoot := filepath.Join(paths.ArtifactsDir, "notarius", "extract-run-1")
|
||||
lanePath := filepath.Join(bundleRoot, "lanes", "encounters.json")
|
||||
indexPath := filepath.Join(bundleRoot, "index.json")
|
||||
trimmedPath := filepath.Join(paths.Root, filepath.FromSlash(artifacts.TranscriptPathFinalTrimmed))
|
||||
mustWriteTestFile(t, lanePath, `{"secret":"DO_NOT_PRINT"}`)
|
||||
mustWriteTestFile(t, indexPath, `{"lanes":[]}`)
|
||||
mustWriteTestFile(t, trimmedPath, `{"segments":[]}`)
|
||||
laneChecksum, err := artifacts.SHA256File(lanePath)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
@@ -1152,11 +1357,22 @@ func writeOperatorExtractionManifest(t *testing.T, workspaceRoot string) string
|
||||
}
|
||||
m := manifest.New("2026-05-03", time.Now().UTC())
|
||||
m.Campaign = "sample-campaign"
|
||||
m.Stages["trim"] = &manifest.StageRecord{
|
||||
Name: "trim", Status: manifest.StatusSucceeded,
|
||||
Outputs: []manifest.ArtifactRecord{{
|
||||
Kind: artifactmodel.TranscriptOutputKindFinalTrimmed, LocalPath: trimmedPath, ProducerRunID: "trim-run-1",
|
||||
}},
|
||||
}
|
||||
input, err := artifacts.ResolveExtractionInputIdentity(paths, m)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m.Stages["extract"] = &manifest.StageRecord{
|
||||
Name: "extract", Status: manifest.StatusSucceeded,
|
||||
Metadata: map[string]any{
|
||||
"narratio_run_id": "extract-run-1", "bundle_root": bundleRoot,
|
||||
"receipt": map[string]any{"run_id": "notarius-run-1", "pipeline_id": "campaign.extract"},
|
||||
"receipt": map[string]any{"run_id": "notarius-run-1", "pipeline_id": "campaign.extract"},
|
||||
"direct_input": input.Metadata(),
|
||||
},
|
||||
Outputs: []manifest.ArtifactRecord{
|
||||
{
|
||||
|
||||
@@ -2,6 +2,7 @@ package app
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path"
|
||||
@@ -12,6 +13,7 @@ import (
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/adapters/storage"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/artifacts"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/previouscache"
|
||||
)
|
||||
|
||||
type stableInputCheck struct {
|
||||
@@ -34,9 +36,9 @@ type remoteAudioCheck struct {
|
||||
}
|
||||
|
||||
type previousArtifactReadiness struct {
|
||||
Requirements []artifacts.PreviousArtifactRequirement
|
||||
MissingID bool
|
||||
Err error
|
||||
Requirements []artifacts.PreviousArtifactRequirement
|
||||
SkippedMissing []string
|
||||
Err error
|
||||
}
|
||||
|
||||
type remoteCurrentStateCheck struct {
|
||||
@@ -51,23 +53,28 @@ type effectiveLocksCheck struct {
|
||||
|
||||
func inspectStableInputs(cfg *config.Config) []stableInputCheck {
|
||||
items := []struct {
|
||||
name string
|
||||
in config.ResolvedInputFile
|
||||
name string
|
||||
in config.ResolvedInputFile
|
||||
optional bool
|
||||
}{
|
||||
{name: "speakers", in: cfg.StableInputs.SpeakersFile},
|
||||
{name: "autocorrect", in: cfg.StableInputs.AutocorrectFile},
|
||||
{name: "glossary", in: cfg.StableInputs.GlossaryFile},
|
||||
{name: "players", in: cfg.StableInputs.PlayersFile},
|
||||
{name: "party", in: cfg.StableInputs.PartyFile},
|
||||
{name: "spell_catalog", in: cfg.StableInputs.SpellCatalogFile, optional: true},
|
||||
}
|
||||
out := make([]stableInputCheck, 0, len(items))
|
||||
for _, item := range items {
|
||||
if item.optional && strings.TrimSpace(item.in.Path) == "" {
|
||||
continue
|
||||
}
|
||||
path, err := resolveHelperConfigRelativePath(item.in)
|
||||
if err != nil {
|
||||
out = append(out, stableInputCheck{Name: item.name, Err: err})
|
||||
continue
|
||||
}
|
||||
if _, err := os.Stat(path); err != nil {
|
||||
if err := requireInspectionFile(path, item.name); err != nil {
|
||||
out = append(out, stableInputCheck{Name: item.name, Path: path, Err: err})
|
||||
continue
|
||||
}
|
||||
@@ -152,23 +159,27 @@ func inspectPreviousArtifactReadiness(
|
||||
if len(requirements) == 0 {
|
||||
return out
|
||||
}
|
||||
if strings.TrimSpace(cfg.Session.PreviousSessionID) == "" {
|
||||
out.MissingID = true
|
||||
if cfg == nil || cfg.Pipeline == nil || cfg.Session == nil {
|
||||
out.Err = fmt.Errorf("resolved config with pipeline/session is required")
|
||||
return out
|
||||
}
|
||||
if store == nil {
|
||||
out.Err = fmt.Errorf("previous-session artifacts cannot be checked because storage is unavailable")
|
||||
paths := artifacts.NewLocalStore(cfg.Pipeline.Workspace.Root).SessionPathsFor(cfg.Session.Campaign, cfg.Session.SessionID)
|
||||
plan, err := previouscache.Resolve(ctx, cfg, paths, requirements, store)
|
||||
if err != nil {
|
||||
var pointerMissing *artifacts.CurrentRunPointerMissingError
|
||||
var manifestMissing *artifacts.CurrentManifestMissingError
|
||||
if errors.As(err, &pointerMissing) {
|
||||
out.Err = fmt.Errorf("remote %w", pointerMissing)
|
||||
return out
|
||||
}
|
||||
if errors.As(err, &manifestMissing) {
|
||||
out.Err = fmt.Errorf("remote %w", manifestMissing)
|
||||
return out
|
||||
}
|
||||
out.Err = fmt.Errorf("remote %w", err)
|
||||
return out
|
||||
}
|
||||
|
||||
prefix := artifacts.S3SessionPrefix(cfg.Pipeline.Storage.S3.RootPrefix, cfg.Session.Campaign, cfg.Session.PreviousSessionID)
|
||||
if _, err := artifacts.LoadCurrentState(ctx, store, prefix, artifacts.CurrentStateValidation{
|
||||
ExpectedSessionID: strings.TrimSpace(cfg.Session.PreviousSessionID),
|
||||
ExpectedCampaign: strings.TrimSpace(cfg.Session.Campaign),
|
||||
ValidateRunID: true,
|
||||
}); err != nil {
|
||||
out.Err = fmt.Errorf("remote %v", err)
|
||||
}
|
||||
out.SkippedMissing = append([]string(nil), plan.SkippedMissing...)
|
||||
return out
|
||||
}
|
||||
|
||||
@@ -265,8 +276,8 @@ func requireInspectionFile(path, label string) error {
|
||||
}
|
||||
return fmt.Errorf("stat %s %q: %w", label, path, err)
|
||||
}
|
||||
if info.IsDir() {
|
||||
return fmt.Errorf("%s %q is a directory", label, path)
|
||||
if !info.Mode().IsRegular() {
|
||||
return fmt.Errorf("%s %q is not a regular file", label, path)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
34
internal/app/operator_inspection_fifo_test.go
Normal file
34
internal/app/operator_inspection_fifo_test.go
Normal file
@@ -0,0 +1,34 @@
|
||||
//go:build unix
|
||||
|
||||
package app
|
||||
|
||||
import (
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"gitea.maximumdirect.net/eric/narratio/internal/config"
|
||||
"golang.org/x/sys/unix"
|
||||
)
|
||||
|
||||
func TestInspectStableInputsRejectsSpellCatalogFIFO(t *testing.T) {
|
||||
root := t.TempDir()
|
||||
sourcePath := filepath.Join(root, "spells.fifo")
|
||||
if err := unix.Mkfifo(sourcePath, 0o644); err != nil {
|
||||
t.Fatalf("Mkfifo() error = %v", err)
|
||||
}
|
||||
|
||||
cfg := &config.Config{StableInputs: config.ResolvedStableInputs{
|
||||
SpellCatalogFile: config.ResolvedInputFile{Path: sourcePath, ConfigPath: filepath.Join(root, "campaign.yml")},
|
||||
}}
|
||||
for _, check := range inspectStableInputs(cfg) {
|
||||
if check.Name != "spell_catalog" {
|
||||
continue
|
||||
}
|
||||
if check.Err == nil || !strings.Contains(check.Err.Error(), "not a regular file") {
|
||||
t.Fatalf("spell catalog check = %#v, want regular-file rejection", check)
|
||||
}
|
||||
return
|
||||
}
|
||||
t.Fatal("spell catalog inspection result was not reported")
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user