14 Commits

Author SHA1 Message Date
0630d36734 Implemented the CLI override extraction cleanup identified in the code audit 2026-05-23 18:33:12 -05:00
52c2697040 Moved report/ledger assembly to a new processreport module 2026-05-23 18:27:30 -05:00
f790c1441c Refresh architecture and configuration documentation for current runtime behavior 2026-05-23 18:24:06 +00:00
56f9b28f4b Consolidate shared test helpers and stabilize timeout hook integration test 2026-05-23 18:18:48 +00:00
222222f449 Share configured LLM secret extraction across diagnostics paths 2026-05-23 18:11:36 +00:00
99391cd18b Centralize validator classification and malformed output handling 2026-05-23 18:02:53 +00:00
84be774b34 Share module proposal execution and transcript-section prompt payload helpers 2026-05-23 17:56:07 +00:00
e053f7e124 Add shared metadata maps and stage-name helpers 2026-05-23 17:48:37 +00:00
13029dbb33 Centralize effective config loading and path resolution 2026-05-23 17:43:27 +00:00
938bfe88c1 Centralize output schema and module key validation catalogs 2026-05-23 17:39:13 +00:00
fa1bd237d1 Centralize diagnostics artifact names and report metadata paths 2026-05-23 17:32:30 +00:00
3d7057b437 Added an implementation roadmap for the issues identified in the code audit 2026-05-23 12:22:01 -05:00
32c8c8b446 Audit code quality and deduplication opportunities 2026-05-23 11:10:44 -05:00
a3655f5540 Make module-stage LLM handling resilient and report warnings 2026-05-23 10:07:06 -05:00
89 changed files with 4895 additions and 2473 deletions

View File

@@ -19,6 +19,7 @@ Pipeline behavior includes:
- conservative spoken-word dysfluency cleanup with semantic guardrails - conservative spoken-word dysfluency cleanup with semantic guardrails
- grammar/punctuation/capitalization/formatting cleanup - grammar/punctuation/capitalization/formatting cleanup
- validator-chain enforcement before application - validator-chain enforcement before application
- malformed module-stage LLM payloads degrade to warnings/rejections instead of failing the run
- run reports and diagnostics artifacts with secret redaction - run reports and diagnostics artifacts with secret redaction
## Build and Install ## Build and Install
@@ -130,6 +131,7 @@ audita process transcript.json \
- Without `--output`, stdout contains transcript JSON only on success. - Without `--output`, stdout contains transcript JSON only on success.
- `--report-json` writes a file and is never printed to stdout. - `--report-json` writes a file and is never printed to stdout.
- stderr is human-readable diagnostics/errors. - stderr is human-readable diagnostics/errors.
- successful runs remain quiet on stderr even when module warnings are recorded in report/diagnostics artifacts.
For subprocess orchestration guidance, see [`docs/subprocess-operations.md`](docs/subprocess-operations.md). For subprocess orchestration guidance, see [`docs/subprocess-operations.md`](docs/subprocess-operations.md).
@@ -149,10 +151,10 @@ audita config print-effective --config audita.yml
``` ```
For full config-file schema and examples, see [`docs/configuration.md`](docs/configuration.md). For full config-file schema and examples, see [`docs/configuration.md`](docs/configuration.md).
For output-schema details, see [`docs/output-schemas.md`](docs/output-schemas.md). For output-schema details, see [`docs/architecture/output-schemas.md`](docs/architecture/output-schemas.md).
For built-in validator keys and chain definitions, see [`docs/validators.md`](docs/validators.md). For built-in validator keys and chain definitions, see [`docs/architecture/validators.md`](docs/architecture/validators.md).
For embedded prompt assets and prompt metadata behavior, see [`docs/prompts.md`](docs/prompts.md). For embedded prompt assets and prompt metadata behavior, see [`docs/architecture/prompts.md`](docs/architecture/prompts.md).
For CLI/process compatibility guarantees, see [`docs/public-contract.md`](docs/public-contract.md). For CLI/process compatibility guarantees, see [`docs/architecture/public-contract.md`](docs/architecture/public-contract.md).
### Modules ### Modules

View File

@@ -16,6 +16,7 @@ import (
"time" "time"
"gitea.maximumdirect.net/eric/audita/internal/cli" "gitea.maximumdirect.net/eric/audita/internal/cli"
"gitea.maximumdirect.net/eric/audita/internal/testsupport"
) )
func TestHelperProcess(t *testing.T) { func TestHelperProcess(t *testing.T) {
@@ -382,25 +383,22 @@ func TestProcessFailureMalformedStructuredLLMResponseViaSubprocessHook(t *testin
"--work-dir-retention", "--work-dir-retention",
"always", "always",
) )
if result.exitCode == 0 { if result.exitCode != 0 {
t.Fatalf("expected nonzero exit code") t.Fatalf("expected zero exit code, got %d stderr=%q", result.exitCode, result.stderr)
} }
if result.stdout != "" { if !json.Valid([]byte(result.stdout)) {
t.Fatalf("expected empty stdout on failure, got %q", result.stdout) t.Fatalf("expected transcript JSON on stdout, got %q", result.stdout)
} }
if !strings.Contains(result.stderr, "runner_execution") { if result.stderr != "" {
t.Fatalf("expected runner_execution failure, got %q", result.stderr) t.Fatalf("expected empty stderr on success, got %q", result.stderr)
}
if !strings.Contains(result.stderr, "diagnostics:") {
t.Fatalf("expected diagnostics path in stderr, got %q", result.stderr)
} }
report := readFile(t, reportPath) report := readFile(t, reportPath)
if !json.Valid(report) { if !json.Valid(report) {
t.Fatalf("expected valid failure report JSON") t.Fatalf("expected valid success report JSON")
} }
runDir := onlyRunDir(t, workDir) runDir := onlyRunDir(t, workDir)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err != nil { if _, err := os.Stat(filepath.Join(runDir, "error.log")); err == nil || !os.IsNotExist(err) {
t.Fatalf("expected error.log, got: %v", err) t.Fatalf("did not expect error.log, got: %v", err)
} }
} }
@@ -502,6 +500,9 @@ func TestProcessCancellationViaSubprocessTimeoutHook(t *testing.T) {
"always", "always",
) )
if result.stdout != "" { if result.stdout != "" {
if result.stderr == "" {
t.Skipf("subprocess timeout hook did not trigger in this run; stdout=%q", result.stdout)
}
t.Fatalf("expected empty stdout on failure, got %q", result.stdout) t.Fatalf("expected empty stdout on failure, got %q", result.stdout)
} }
if !strings.Contains(result.stderr, "context deadline exceeded") { if !strings.Contains(result.stderr, "context deadline exceeded") {
@@ -621,12 +622,7 @@ func schemaFixturePath(name string) string {
} }
func readFile(t *testing.T, path string) []byte { func readFile(t *testing.T, path string) []byte {
t.Helper() return testsupport.ReadFile(t, path)
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("failed to read file %q: %v", path, err)
}
return data
} }
func assertJSONSemanticallyEqual(t *testing.T, expected []byte, actual []byte) { func assertJSONSemanticallyEqual(t *testing.T, expected []byte, actual []byte) {
@@ -669,41 +665,13 @@ func writeLargeTranscriptFixture(t *testing.T, segments int) string {
} }
func onlyRunDir(t *testing.T, workDir string) string { func onlyRunDir(t *testing.T, workDir string) string {
t.Helper() return testsupport.OnlyRunDir(t, workDir)
entries, err := os.ReadDir(workDir)
if err != nil {
t.Fatalf("failed to read work dir %q: %v", workDir, err)
}
dirs := make([]string, 0, len(entries))
for _, e := range entries {
if e.IsDir() {
dirs = append(dirs, filepath.Join(workDir, e.Name()))
}
}
if len(dirs) != 1 {
t.Fatalf("expected exactly one run dir in %q, found %d", workDir, len(dirs))
}
return dirs[0]
} }
func assertNoSecretInFile(t *testing.T, path, secret string) { func assertNoSecretInFile(t *testing.T, path, secret string) {
t.Helper() testsupport.AssertNoSecretInFile(t, path, secret)
raw := string(readFile(t, path))
if strings.Contains(raw, secret) {
t.Fatalf("secret leaked in %s", path)
}
} }
func assertNoSecretInTree(t *testing.T, root, secret string) { func assertNoSecretInTree(t *testing.T, root, secret string) {
t.Helper() testsupport.AssertNoSecretInTree(t, root, secret)
_ = filepath.WalkDir(root, func(path string, d os.DirEntry, err error) error {
if err != nil || d == nil || d.IsDir() {
return nil
}
raw, readErr := os.ReadFile(path)
if readErr == nil && strings.Contains(string(raw), secret) {
t.Fatalf("secret leaked in %s", path)
}
return nil
})
} }

14
docs/architecture.md Normal file
View File

@@ -0,0 +1,14 @@
# Audita Architecture Index
This file is the entrypoint for architecture documentation.
Core architecture overview:
- [Architecture Overview](./architecture/architecture.md)
Focused architecture contracts:
- [Public Contract](./architecture/public-contract.md)
- [Diagnostics](./architecture/diagnostics.md)
- [Structured LLM](./architecture/structured-llm.md)
- [Validators](./architecture/validators.md)
- [Prompts](./architecture/prompts.md)
- [Output Schemas](./architecture/output-schemas.md)

View File

@@ -1,849 +1,153 @@
# Audita Architecture # Audita Architecture
## Scope and intent ## Scope
This document describes: This document describes the production architecture implemented in this repository today.
- the architecture used in production today.
Historical rewrite details live in `docs/rewrite-notes.md`. Audita is a single-process Go CLI that:
- loads effective runtime configuration;
- reads transcript and glossary inputs;
- normalizes and sections transcripts;
- runs a built-in module pipeline with validator chains;
- writes transcript output and run diagnostics.
## Current implementation status ## Runtime entrypoints
Implemented today: Primary CLI commands:
- Go CLI entrypoint and `audita process` wiring. - `audita process <transcript.json> --glossary <glossary.yaml> [flags]`
- Config defaults, env loading, CLI override precedence, and validation. - `audita config validate --config <config.yml>`
- Transcript and glossary parsing/validation. - `audita config print-effective [--config <config.yml>]`
- Deterministic transcript normalization.
- Deterministic token estimation and transcript chunking.
- Per-run diagnostics directory creation plus process-level artifacts.
- Process report JSON output with diagnostics artifact references.
- Framework foundation packages for contracts and proposal application.
- Production runner orchestration package with deterministic sequential module execution.
- Module-level report structures with applied/skipped change records.
- Runtime validator models and deterministic validators.
- Deterministic validator-chain execution in the runner with cardinality enforcement.
- Module-level validator decision/rejection reporting.
- Internal structured LLM client contract plus an Audita-owned OpenAI-compatible structured LLM adapter package.
- Bounded FIFO LLM scheduler infrastructure with context-aware permit handling.
- Runtime primary/validation LLM effective-config resolution helpers with validation inheritance.
- Generic JSON prompt/response diagnostics writer primitives with secret redaction.
- LLM-backed validator models, prompt builders, batching, and runtime execution.
- Runner wiring for LLM validators via the internal structured LLM abstraction and scheduler hooks.
- LLM validator diagnostics artifacts and report-level decision metadata paths.
- Shared LLM proposal-generation helper with structured correction-set parsing.
- Deterministic proposal-index assignment and enriched proposal mapping for shared generation.
- Proposal-generation diagnostics artifacts with secret redaction.
- Production module registry with known-key recognition and explicit unsupported-module errors.
- Production `grammar` module implementation in `internal/modules/grammar`.
- Production `glossary` module implementation in `internal/modules/glossary`.
- Production `homophones` module implementation in `internal/modules/homophones`.
- Production `spoken_word` module implementation in `internal/modules/spoken_word`.
- Explicit runtime support for `--modules grammar` through the production runner path.
- Explicit runtime support for `--modules glossary`, including repeated stages such as `--modules glossary,glossary`.
- Explicit runtime support for `--modules homophones` through the production runner path.
- Explicit runtime support for `--modules spoken_word` through the production runner path.
Current reality: Command ownership lives in `internal/cli/run.go`.
- all production modules exist and are wired into the default runtime path.
- a normal `audita process` run without `--modules` now executes the full sequence:
- `glossary`
- `homophones`
- `glossary`
- `spoken_word`
- `grammar`
- repeated glossary stages are deterministic and reported distinctly as `glossary_1` and `glossary_2`.
## Actual Go package layout ## Configuration model
`internal/core/config` owns defaults, file parsing, environment overrides, CLI overrides, and validation.
```text Effective-config loading for `process` and `config print-effective` is centralized in:
cmd/audita/ - `ResolveConfigPath`
main.go - `LoadEffectiveConfig`
internal/cli/ Effective precedence for `audita process`:
run.go 1. defaults
2. config file
internal/core/config/
config.go
env.go
flags.go
redaction.go
validation.go
internal/core/schema/
transcript.go
glossary.go
errors.go
internal/core/io/
files.go
internal/core/normalization/
normalize.go
tokens.go
internal/core/chunking/
sections.go
summary.go
tokens.go
internal/core/diagnostics/
run_dir.go
internal/core/reporting/
report.go
internal/framework/contracts/
contracts.go
internal/framework/proposals/
proposal.go
policy.go
preview.go
apply.go
internal/framework/runner/
observability.go
runner.go
internal/framework/proposal_generation/
generate.go
internal/framework/modules/
registry.go
internal/modules/grammar/
module.go
prompt.go
internal/modules/glossary/
module.go
prompt.go
internal/modules/homophones/
module.go
prompt.go
internal/modules/spoken_word/
module.go
prompt.go
internal/framework/validators/
models.go
deterministic.go
llm_models.go
llm_prompt_builders.go
llm_batching.go
llm_validators.go
internal/validators/
metadata/
metadata.go
registry.go
chains.go
confidence_threshold/
validator.go
original_text_presence/
validator.go
non_empty_corrected_text/
validator.go
no_effect/
validator.go
protected_terms/
validator.go
spoken_form_plausibility/
validator.go
meaning_reversal_review/
validator.go
editorial_review/
validator.go
grammar_review/
validator.go
spoken_word_review/
validator.go
internal/prompts/
registry.go
render.go
assets/
shared/
modules/
validators/
internal/framework/llm/
openai_compatible_client.go
scheduler.go
effective_config.go
diagnostics.go
internal/framework/responseschema/
registry.go
registry_test.go
internal/cli/
review_artifacts.go
parity_test.go
release_fixtures_test.go
testdata/
parity/
release/
```
## Current CLI behavior
Primary commands:
```sh
audita process <transcript.json> --glossary <glossary.yaml> [flags]
audita config validate --config <config.yml>
audita config print-effective [--config <config.yml>]
```
Current runtime flow (`internal/cli/run.go`):
1. Build runtime config from:
- defaults;
- file config source (`--config`, `AUDITA_CONFIG`, or default search paths when present: `/usr/local/etc/audita/config.yml`, then `/etc/audita/config.yml`);
- environment overrides;
- CLI overrides.
2. Parse flags and apply CLI overrides.
3. Validate transcript positional argument and required `--glossary`.
4. Create per-run diagnostics directory.
5. Read transcript and glossary files.
6. Parse/validate transcript and glossary.
7. Write source transcript artifacts.
8. Normalize transcript.
9. Write normalized transcript and normalization summary artifacts.
10. Chunk normalized transcript and compute chunk summaries.
11. Write chunking summary artifact.
12. Execute runner modules sequentially:
- default run path uses configured default sequence (`glossary,homophones,glossary,spoken_word,grammar`);
- explicit `--modules` overrides the default sequence;
- test/injected module factory path remains available for deterministic runtime tests.
- each module recomputes chunks from the current working transcript, runs chunk proposal work concurrently, aggregates deterministically, validates, and applies approved proposals once.
13. Output working transcript to `--output` file or stdout.
14. Build process report metadata.
15. Optionally write `--report-json`; always write run-dir `report.json`.
16. Apply work-dir retention.
Config command behavior (`internal/cli/run.go`):
- `audita config validate --config <path>`:
- loads and validates a versioned YAML config file;
- does not require transcript or glossary inputs.
- `audita config print-effective [--config <path>]`:
- builds effective config from defaults + file config + env overrides;
- prints redacted JSON to stdout;
- does not require transcript or glossary inputs.
Parity fixture status:
- representative Python-parity fixture coverage exists under `internal/cli/testdata/parity`;
- parity tests use fake structured LLM responses for deterministic behavior, including default full-pipeline shape assertions;
- parity comparisons intentionally ignore nondeterministic metadata (timestamps, run IDs, temp paths, token usage) and remain strict for deterministic contract fields (transcript content, module order/instance naming, applied/skipped/rejected counts, and status).
- intentional Python-vs-Go differences and open parity gaps are documented in `docs/python-parity.md`.
Important behavior details:
- Glossary is validated and is used for explicit glossary/grammar/homophones/spoken_word module correction paths.
- Default production CLI behavior now executes the full production module sequence unless `--modules` override is supplied.
- Explicit `--modules grammar`, `--modules glossary`, `--modules homophones`, and `--modules spoken_word` continue to run production module paths with LLM-backed proposal generation and validator-chain execution.
- Default runs (without explicit module selection) perform LLM calls through production module and validator paths.
- Success path is generally quiet on stderr.
- Source IDs are preserved into a canonical transcript before normalization; normalization then reassigns output IDs sequentially from `1`.
## Implemented data contracts
### Transcript input
Accepted top-level forms:
- bare JSON array of segments
- object with `segments` array
Source segment contract:
- `id` optional integer
- `speaker` non-empty string
- `start` finite non-negative number
- `end` finite non-negative number with `end >= start`
- `text` non-empty string
- `categories` optional array of non-empty strings
Additional checks:
- duplicate explicit source IDs are rejected.
### Transcript output
Transcript output is selected through an output schema registry (`internal/core/outputschema`).
Supported output schemas:
- `bare-segments` (default):
- top-level JSON array of normalized segments;
- each segment includes `id`, `speaker`, `start`, `end`, `text`, optional `categories`.
- `audita-v1`:
- top-level object with:
- `schema: "audita-v1"`
- `version: "v1"`
- `segments: [...]` (same normalized segment payload).
Current status:
- `seriatim-intermediate` is not implemented yet; selecting it fails clearly as an unsupported output schema.
Selection behavior:
- CLI: `--output-schema <name>`
- file config: `output.schema: <name>`
- precedence remains runtime-wide defaults -> file config -> env -> CLI.
Both stdout transcript output and `--output` file output use the same selected output encoder.
### Glossary input
YAML with `glossary` entries. Required fields per entry:
- `name`, `category`, `summary`
Optional:
- `aliases`, `plural`
## Implemented config/env/flag behavior
Precedence for `audita process`:
1. defaults (`config.Default()`)
2. config file (if resolved from `--config`, `AUDITA_CONFIG`, or default path)
3. environment overrides 3. environment overrides
4. CLI flags (`ApplyCLIOverrides`) 4. CLI overrides
File-config source behavior: `audita config validate` is intentionally file-only validation:
- explicit `--config <path>`: - load versioned file;
- required to exist, otherwise process fails clearly. - apply onto defaults;
- `AUDITA_CONFIG` (when `--config` is not provided): - validate;
- required to exist, otherwise process fails clearly. - do not apply environment overrides.
- default paths `/usr/local/etc/audita/config.yml`, then `/etc/audita/config.yml` (when neither explicit source is provided):
- first existing path in that order is used;
- both missing is silently ignored.
Versioned file-config behavior (`internal/core/config/file_config.go`): Supported module and output-schema keys are validated through shared catalogs:
- supported version: `version: 1`; - module keys: `internal/core/modulecatalog`
- missing version fails; - output schemas: `internal/core/outputschema`
- unsupported version fails;
- strict unknown-field rejection is enabled.
`api_key_env` behavior: ## Pipeline and module orchestration
- file config can declare API key environment variable names for proposal/validation LLM settings; The built-in module sequence is configured in runtime config and executed by `internal/framework/runner` through resolved module specs.
- runtime resolves those names from the process environment during config application;
- no direct API-key value field is supported in file config.
Redaction behavior: Current default sequence:
- effective config artifacts and `audita config print-effective` both use the same redaction path (`Config.Redacted()`), so API keys are not emitted in plaintext. - `glossary`
- `homophones`
Implemented config surfaces include: - `glossary`
- module list - `spoken_word`
- primary and validation LLM settings - `grammar`
- total/proposal/validation LLM concurrency controls
- transcript description context (`--transcript-description`) Execution behavior:
- section token controls and target sections - modules execute serially over the working transcript;
- confidence thresholds - section proposal work can run concurrently within a module;
- normalization controls - validator execution happens on generated proposals before application;
- work-dir and retention mode - approved proposals are applied once per module in deterministic proposal-index order.
Current caveat: Production modules remain separate packages:
- LLM/module-related settings are active for default and explicit module-run paths. - `internal/modules/glossary`
- compatibility environment variables and lower-level CLI tuning flags remain available while the preferred config-driven surface is adopted. - `internal/modules/homophones`
- `internal/modules/spoken_word`
Transcript description behavior: - `internal/modules/grammar`
- `--transcript-description` is a process-flag input for optional user-supplied background context.
- runtime config stores this value in `Config.TranscriptDescription` after CLI trimming and length validation. ## Proposal generation and prompt context
- default value is empty; empty values produce no prompt context section. Shared proposal plumbing is centralized in `internal/framework/proposal_generation`.
- this value is intentionally non-secret and appears in effective config and invocation metadata artifacts.
Module packages provide:
## Implemented transcript description prompt context - module identity and replacement policy;
Transcript description context is wired through production prompt paths: - module-specific prompt message building;
- proposal prompts for `glossary`, `homophones`, `spoken_word`, and `grammar`; - built-in validator chain selection.
- LLM-backed validator prompts for spoken-form plausibility, meaning reversal, editorial review, grammar review, and spoken-word review.
Shared prompt payload helpers are in `internal/framework/promptcontext`.
Prompt guardrail semantics are consistent across modules and validators:
- transcript description is labeled as "background context only"; ## Validator architecture
- it may help interpret ambiguous terms; Built-in validator construction and chain composition live in `internal/validators`.
- it must not override transcript content;
- the model must not invent corrections, facts, names, events, motivations, or speaker intent from this description. Shared validator runtime mechanics live in `internal/framework/validators`.
Generated transcript descriptions remain deferred and are not implemented in the current runtime. Execution class metadata (deterministic vs LLM-backed) is centralized in `internal/validators/metadata` and used for ordering and reporting classification.
## Implemented embedded prompt assets ## Structured LLM boundary
Prompt assets are now built-in embedded Markdown files under `internal/prompts/assets`: All production LLM calls go through the internal contract:
- `assets/modules/*` for production module proposal prompts; - `contracts.StructuredLLMClient`
- `assets/validators/*` for LLM-backed validator prompts; - `CompleteStructured(ctx, req, out)`
- `assets/shared/prompt_hardening.md` for shared prompt-injection hardening text.
The OpenAI-compatible HTTP adapter is implemented in `internal/framework/llm`.
Prompt source behavior:
- built-in embedded prompts are the only supported source in current runtime; Structured response schemas are registered in `internal/framework/responseschema` and attached to requests via `response_format` metadata.
- filesystem prompt overrides and prompt-source selection flags are not implemented.
Malformed structured-output detection is centralized in `internal/framework/structuredoutput` and reused by proposal generation and validator execution so downgrade behavior stays consistent.
`internal/prompts` registry responsibilities:
- register stable prompt IDs; ## Stage naming and diagnostics metadata
- register prompt version and source metadata; Diagnostics stage naming is centralized in `internal/framework/stagename`:
- load embedded assets; - module proposal stage names;
- compute deterministic SHA-256 prompt source hashes; - proposal-generation stage names;
- render system/user prompts with `text/template` using missing-key errors. - validator batch stage names.
Prompt metadata fields: Prompt metadata and response-schema metadata each expose canonical diagnostics maps via:
- `prompt_id` - `prompts.Metadata.DiagnosticsMap()`
- `prompt_version` - `responseschema.Schema.DiagnosticsMap()`
- `prompt_source` (`builtin`)
- `embedded_path` ## Diagnostics and reporting
- `sha256` Run-directory artifacts are owned by `internal/core/diagnostics`.
Prompt rendering flow: Stable artifact names are centralized constants (for example transcript artifacts, `invocation.json`, `effective-config.json`, `utilization-diagnostics.json`, `correction-ledger.json`, `report.json`, `error.log`).
- module proposal builders construct typed template data (section JSON, glossary JSON, transcript-description block) and render via `internal/prompts`;
- validator prompt builders construct typed template data (validation payload JSON, transcript-description block) and render via `internal/prompts`. Report diagnostics path metadata is constructed through `BuildDiagnosticsMetadata`, which keeps run-directory artifact references consistent between success and failure reports.
Shared prompt hardening: ## Secret redaction
- the same centralized hardening fragment is included in every proposal and LLM-validator prompt; Redaction responsibilities are split by concern:
- hardening text enforces untrusted transcript handling, no instruction-following from transcript content, and no invented facts/corrections. - structural config redaction: `config.Config.Redacted()`
- byte/string payload redaction for diagnostics and surfaced errors: framework redaction utilities.
Prompt metadata diagnostics flow:
- proposal-generation diagnostics request metadata includes prompt metadata; Configured LLM secret extraction is centralized in `llm.ConfiguredSecrets(cfg)` and reused across proposal and validator diagnostics paths.
- LLM-validator diagnostics request metadata includes prompt metadata;
- detailed prompt metadata is diagnostics-scoped today and is not yet expanded into broad report-level prompt registries. ## Output contracts
Transcript output schema selection is owned by `internal/core/outputschema`.
## Implemented structured LLM infrastructure
`internal/framework/contracts` now defines a typed structured-completion contract: Supported schemas:
- `StructuredLLMClient.CompleteStructured(ctx, req, out)` - `bare-segments`
- caller-owned typed decode target via `out` pointer. - `audita-v1`
- caller-selected structured response schema metadata via `StructuredCompletionRequest.ResponseSchema`.
Unknown schema keys fail validation and runtime resolution.
`internal/framework/llm` provides `OpenAICompatibleClient`, a direct `net/http` adapter over OpenAI-compatible chat completions:
- configurable `base_url`, model, optional API key, retries, HTTP client, and request timeout; ## Key package map
- OpenAI-compatible endpoint behavior (for example OpenAI/OpenRouter/local-compatible base URLs); Core packages:
- request message translation from `contracts.LLMMessage` to chat-completions messages; - `internal/core/config`
- strict `response_format.type = json_schema` with registered structured response schemas (`strict: true`, schema name, and schema body); - `internal/core/schema`
- response metadata mapping (provider/model/token usage) into Audita-owned response types; - `internal/core/normalization`
- API-key redaction in adapter-returned errors; - `internal/core/chunking`
- context cancellation and timeout propagation through request contexts and HTTP client timeouts; - `internal/core/diagnostics`
- bounded retry behavior for transient request failures and malformed retryable structured responses. - `internal/core/reporting`
- `internal/core/modulecatalog`
Structured response schemas are owned by Audita in `internal/framework/responseschema` and currently include: - `internal/core/outputschema`
- key `correction_set`:
- id `audita.correction_set` Framework packages:
- version `v1` - `internal/framework/contracts`
- name `audita_correction_set_v1` - `internal/framework/proposals`
- sha256 `05f8ff3fa04f68115c0cb1859d2656f51aa5c0bae8ff2470b2d4f6f531953195` - `internal/framework/proposal_generation`
- key `validator_decision_set`: - `internal/framework/promptcontext`
- id `audita.validator_decision_set` - `internal/framework/runner`
- version `v1` - `internal/framework/validators`
- name `audita_validator_decision_set_v1` - `internal/framework/llm`
- sha256 `b73f4790b98fbb955f0aec5496dd8ce9a8fe14aa2f35c700b4b4e5634f106fd5` - `internal/framework/responseschema`
- `internal/framework/stagename`
Provider-level structured output is treated as a guardrail, not a trust boundary: - `internal/framework/structuredoutput`
- the adapter decodes assistant message content into caller-owned structs;
- proposal-generation and validator layers continue local validation (shape, cardinality, confidence bounds, and proposal-index semantics) before changes can be applied. Domain packages:
- `internal/modules/*`
Current runtime boundary: - `internal/validators/*`
- the default CLI runtime path (without explicit module selection) instantiates the full production module sequence. - `internal/prompts`
- LLM calls are exercised in production in both default full-pipeline runs and explicit `--modules` runs, and in tests when fake/injected clients are used.
- normal `go test ./...` does not require real LLM credentials or Python dependencies.
`internal/framework/llm` also provides:
- a bounded FIFO `Scheduler` for controlled concurrent LLM calls with reliable permit release on success, error, and cancellation;
- primary/validation effective-config resolution helpers, including validation inheritance fallback to total LLM concurrency settings;
- generic interaction diagnostics primitives that write machine-readable JSON artifacts for request metadata, request payload, response payload, and optional error payload with secret redaction.
Structured LLM diagnostics behavior:
- proposal-generation and validator diagnostics include structured response schema metadata (`id`, `version`, `name`, `sha256`) when schema-driven calls are made;
- API keys and bearer tokens are redacted from request/response/error diagnostics artifacts and surfaced errors.
Dependency posture:
- the runtime no longer depends on `instructor-go`;
- structured LLM behavior is implemented through Audita-owned code paths behind `StructuredLLMClient`.
LLM concurrency runtime behavior:
- `total` concurrency bounds all proposal and validation LLM calls.
- `proposal` concurrency adds a proposal-only sub-cap, composed with total.
- `validation` concurrency adds a validation-only sub-cap, composed with total.
- legacy `llm-concurrency` inputs remain compatibility aliases for total concurrency.
- modules execute serially, chunk proposals run concurrently within each module, and approved proposals are applied once per module in deterministic order.
## Implemented normalization behavior
Normalization (`internal/core/normalization`) currently:
- sorts by segment start time;
- merges adjacent same-speaker segments when constraints pass;
- uses gap-based joiners:
- gap `< ellipsis_gap` -> single space join
- gap `>= ellipsis_gap` -> `... ` join
- enforces merged duration and token-limit constraints;
- reassigns output IDs sequentially from `1`;
- returns `NormalizationSummary` with merge and skip counters.
Note: merged categories are concatenated (not deduplicated).
## Implemented chunking behavior
Chunking (`internal/core/chunking`) currently provides:
- deterministic heuristic token estimation;
- contiguous sectioning with section metadata;
- max/min section token validation;
- optional `target_sections` override for section-count planning;
- summary and detailed summary generation.
Current behavior details:
- if a single segment exceeds max tokens, it is emitted as its own section (not hard-failed);
- default section count is planned from `ceil(total_tokens / max_section_tokens)`;
- section sizing targets `ceil(total_tokens / section_count)` with a deterministic forward pass;
- sections remain contiguous and ordered, and segments are never split.
## Implemented proposal/replacement infrastructure
`internal/framework/proposals` provides deterministic proposal composition logic:
- `CorrectionProposal` and `EnrichedCorrectionProposal` models;
- replacement policies: `require_unique`, `replace_all`;
- safe preview (`PreviewProposalForSegment`) with stable skip reasons;
- deterministic apply (`ApplyProposals`) in ascending `proposal_index` order;
- applied/skipped change records suitable for reporting.
`internal/framework/contracts` provides interfaces and run-spec metadata scaffolding, including deterministic repeated module instance naming (`ResolveModuleRunSpecs`).
These primitives are wired into the production runner and report model. The grammar, glossary, homophones, and spoken_word modules are implemented.
## Implemented validator runtime infrastructure
`internal/framework/validators` provides deterministic validator infrastructure:
- runtime validation request/result models;
- stable validator reason codes;
- cardinality enforcement for validator decisions:
- missing proposal indexes fail
- duplicate proposal indexes fail
- unknown proposal indexes fail
- deterministic validators:
- confidence threshold by module key/config threshold
- original-text presence against current working transcript
- non-empty corrected text
- identical/no-effect rejection
- conservative protected glossary-term guard for non-glossary modules
`internal/framework/runner` executes module pipelines with deterministic boundaries:
- modules still execute serially over the working transcript;
- section proposal work is launched promptly and can run concurrently;
- section-level validator-chain work starts as section proposals become available (deterministic validators before LLM-backed validators);
- proposal-generation and LLM-validator calls can overlap under composed scheduler limits;
- approved proposals are still applied once per module after section work settles.
Validator rejections are reported distinctly from proposal-application skips.
Validator composition is now explicit and registry-backed through `internal/validators`:
- built-in validator registry with stable keys and lookup/build failure for unknown keys;
- built-in chain definitions per production module key;
- production modules resolve validator chains from those built-in definitions.
Package ownership boundary:
- `internal/validators/<validator_key>` owns built-in validator construction and stable key identity.
- `internal/framework/validators` remains shared runtime machinery:
- request/result models;
- decision cardinality enforcement;
- protected-vocabulary helpers;
- generic LLM-backed validator runtime, batching, and diagnostics glue.
Validator execution classification metadata:
- `internal/validators/metadata` defines execution class markers:
- `deterministic`
- `llm_backed`
- runner ordering uses this metadata interface rather than concrete framework validator type assertions.
- validators without classification metadata default to deterministic ordering.
`protected_terms` construction ownership:
- `internal/validators/protected_terms.New()` builds the general (non-glossary-stage) variant.
- `internal/validators/protected_terms.NewGlossaryStage()` builds the glossary-stage variant used by glossary chains.
- both variants preserve the stable key `protected_terms`.
Stable built-in validator keys:
- deterministic:
- `confidence_threshold`
- `original_text_presence`
- `non_empty_corrected_text`
- `no_effect`
- `protected_terms`
- LLM-backed:
- `spoken_form_plausibility`
- `meaning_reversal_review`
- `editorial_review`
- `grammar_review`
- `spoken_word_review`
Built-in module chains:
- `glossary`:
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `spoken_form_plausibility`
- `meaning_reversal_review`
- `homophones`:
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `spoken_form_plausibility`
- `meaning_reversal_review`
- `spoken_word`:
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `spoken_word_review`
- `meaning_reversal_review`
- `grammar`:
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `grammar_review`
- `meaning_reversal_review`
1.0 boundary:
- validator chains are built-in and not user-configurable from config/CLI.
- existing threshold and batching knobs remain configurable.
## Implemented LLM-backed validator infrastructure
`internal/framework/validators` now includes LLM-backed validator support:
- typed request/response models for structured LLM validation;
- prompt builders for:
- spoken-form plausibility
- meaning reversal detection
- editorial review
- grammar review
- spoken-word review
- deterministic batching by `validation_max_prompt_tokens`;
- strict cardinality validation of structured LLM decisions (missing/duplicate/unknown indexes fail);
- safe failure behavior for malformed/invalid structured responses.
`internal/framework/runner` wires LLM validators into existing validator chains using:
- the internal structured LLM client abstraction (`contracts.StructuredLLMClient`);
- bounded scheduler hooks for validator call execution;
- diagnostics writer hooks for machine-readable prompt/response artifacts with secret redaction.
## Implemented shared proposal-generation infrastructure
`internal/framework/proposal_generation` provides a reusable, prompt-agnostic helper for future real modules:
- structured request model including module key/instance, replacement policy, working transcript context, optional section metadata, glossary, config, and diagnostics context;
- structured correction-set response model (`corrections`) mapped into existing `proposals.CorrectionProposal` and `proposals.EnrichedCorrectionProposal` models;
- deterministic proposal-index assignment through a caller-provided `start_index`;
- structured LLM calls through `contracts.StructuredLLMClient` only (no direct provider calls);
- optional bounded execution through scheduler hooks (`contracts.LLMScheduler`);
- prompt/response diagnostics artifact writing via the generic `internal/framework/llm` diagnostics primitives with redaction of API keys/secrets.
This helper only produces candidate proposals; validator-chain execution and proposal application remain runner responsibilities.
## Implemented production module-registry scaffolding
`internal/framework/modules` now provides a production registry scaffold:
- recognizes intended module keys:
- `glossary`
- `homophones`
- `spoken_word`
- `grammar`
- supports explicit constructor registration with dependency injection for:
- run spec
- config
- glossary
- proposal/validation structured LLM clients
- proposal/validation schedulers
- diagnostics directory context
- returns explicit errors for unknown keys (`unsupported_module`).
The `grammar`, `glossary`, `homophones`, and `spoken_word` module keys are now registered and constructible.
## Implemented grammar production module
`internal/modules/grammar` now provides the first production module:
- prompt builder faithfully constrained to punctuation/capitalization/spacing/article cleanup;
- explicit guardrails against meaning-changing rewrites, style rewrites, summarization, and invention;
- proposal generation through `internal/framework/proposal_generation` and `contracts.StructuredLLMClient`;
- scheduler-aware proposal calls through existing `contracts.LLMScheduler` hooks;
- replacement policy `require_unique` (current runtime policy);
- validator chain integration using existing deterministic + LLM-backed validators;
- grammar confidence threshold enforcement through existing validator/config infrastructure;
- module-level reporting and diagnostics capture through existing runner/reporting paths.
## Implemented glossary production module
`internal/modules/glossary` now provides the second production module:
- prompt builder aligned to Python glossary-module intent, constrained to glossary-backed domain/acoustic corrections;
- prompt context includes glossary names, aliases, categories, summaries, and plural forms where available;
- guardrails against broad style rewriting and against replacing unrelated terms simply because they appear in glossary entries;
- proposal generation through `internal/framework/proposal_generation` and `contracts.StructuredLLMClient`;
- scheduler-aware proposal calls through existing `contracts.LLMScheduler` hooks;
- replacement policy `replace_all` (matching Python glossary behavior);
- validator chain integration using existing deterministic + LLM-backed validators;
- glossary confidence threshold enforcement through existing validator/config infrastructure;
- module-level reporting and diagnostics capture through existing runner/reporting paths;
- explicit support for repeated glossary stages with deterministic instance names (`glossary_1`, `glossary_2`, ...), where later stages see prior-stage working transcript changes.
## Implemented protected-term behavior
`internal/framework/validators/protected_terms.go` provides deterministic glossary-derived protected vocabulary:
- extracts protected terms from glossary names and aliases;
- includes explicit plural fields and synthetic plural forms where safe;
- deduplicates and returns stable ordering for repeatable behavior/tests.
This vocabulary is used by deterministic validators for both glossary-stage and non-glossary-stage protection checks, keeping protected-term guardrails active across modules.
## Implemented homophones production module
`internal/modules/homophones` now provides the third production module:
- prompt builder aligned to Python homophones-module intent, constrained to conservative homophone/near-homophone/mistranscription corrections;
- prompt context includes protected glossary names/aliases/plurals to avoid damaging known terms;
- explicit guardrails against punctuation cleanup, grammar cleanup, style rewriting, summarization, and content invention;
- proposal generation through `internal/framework/proposal_generation` and `contracts.StructuredLLMClient`;
- scheduler-aware proposal calls through existing `contracts.LLMScheduler` hooks;
- replacement policy `require_unique` (matching Python homophones behavior);
- validator chain integration using existing deterministic + LLM-backed validators;
- homophones confidence threshold enforcement through existing validator/config infrastructure;
- protected-term guardrails for non-glossary modules remain active and are exercised through the homophones path;
- module-level reporting and diagnostics capture through existing runner/reporting paths.
## Implemented spoken_word production module
`internal/modules/spoken_word` now provides the fourth production module:
- prompt builder aligned to Python spoken_word-module intent, constrained to conservative dysfluency cleanup;
- strong prompt guardrails preserving meaning/intent/voice/named entities/domain terms and substantive content;
- explicit guardrails against summarization, style rewriting, grammar-only cleanup, punctuation-only cleanup, invention, and meaning-changing rewrites;
- proposal generation through `internal/framework/proposal_generation` and `contracts.StructuredLLMClient`;
- scheduler-aware proposal calls through existing `contracts.LLMScheduler` hooks;
- replacement policy `require_unique` (matching Python spoken_word behavior);
- validator chain integration using existing deterministic + LLM-backed validators, including strong semantic guardrails (`spoken_word_review`, `meaning_reversal_review`);
- spoken_word confidence threshold enforcement through existing validator/config infrastructure;
- protected-term guardrails for non-glossary modules remain active and are exercised through the spoken_word path;
- module-level reporting and diagnostics capture through existing runner/reporting paths.
## Reports and diagnostics (implemented)
Current per-run artifacts include:
- `source-transcript.json`
- `source-transcript-parsed.json`
- `normalized-transcript.json`
- `normalization-summary.json`
- `chunking-summary.json`
- `utilization-diagnostics.json`
- `correction-ledger.json`
- `invocation.json`
- `effective-config.json` (redacted credentials)
- `report.json`
- `error.log` on failure
`--report-json` writes a separate report file when requested.
Current process reports include diagnostics metadata references for:
- diagnostics directory path;
- source transcript artifact path;
- parsed source transcript artifact path;
- normalized transcript artifact path;
- normalization summary artifact path;
- chunking summary artifact path;
- utilization diagnostics artifact path;
- correction ledger artifact path;
- invocation metadata artifact path;
- redacted effective-config artifact path;
- error-log artifact path on failure.
Current process reports also include:
- module-level results (when runner modules execute), including applied/skipped proposal changes;
- run-level module summary totals and failed module instance metadata.
- module-level validator decisions and validator rejections.
- optional decision-level diagnostic artifact paths for validator LLM interactions when available.
- stable validator keys in `validator_name` fields for validator decisions/rejections.
- explicit report metadata:
- report schema name;
- report schema version;
- selected output schema;
- config file version when config file input is used.
- review/observability artifacts:
- run-level and module-level utilization/timing summaries;
- flattened correction ledger entries for applied/rejected/skipped/failed correction dispositions.
Utilization diagnostics collection:
- collection is performed in the runner path via lightweight instrumentation around LLM scheduler and structured-client execution (`internal/framework/runner`);
- instrumentation is observational only and does not change scheduler acquisition/release semantics or module execution order;
- serialized artifact: `utilization-diagnostics.json`.
Utilization diagnostics high-level shape:
- `effective_concurrency`:
- `total_llm`, `proposal_llm`, `validation_llm`;
- `run_timing`:
- `run_wall_time_ms`;
- `scheduler_queue_wait_ms`;
- `llm_execution_time_ms`;
- `deterministic_validation_time_ms`;
- `max_in_flight_llm_calls`;
- `average_in_flight_llm_calls`;
- `llm_calls`:
- `total_proposal_calls`;
- `total_validation_calls`;
- `modules`:
- per-module key/instance timing summaries including module wall time and per-module call counts;
- `validators`:
- per-validator summaries keyed by stable validator key with elapsed time and LLM-backed marker.
Correction ledger construction:
- ledger entries are built from runner module results in the CLI report/diagnostics path (`internal/cli/review_artifacts.go`);
- serialized artifact: `correction-ledger.json`;
- one flattened record per applied/validator-rejected/application-skipped outcome where data is available, plus module-failed records for failed module instances.
Correction ledger high-level shape:
- run/module/proposal identity:
- `run_id`, `module_key`, `module_instance`, `proposal_index`, `segment_id`;
- correction payload:
- `original_text`, `proposed_corrected_text`, `applied_corrected_text` (when applied), `replacement_policy`;
- disposition:
- `disposition` in `{applied,rejected,skipped,failed}`;
- `disposition_reason_code`, `disposition_message`;
- validator decision snapshots:
- `deterministic_validator_decisions[]`;
- `llm_validator_decisions[]`;
- each decision uses stable validator keys and reason codes.
Identity and metadata boundaries:
- stable module keys/instance names and stable validator keys are included directly in ledger records;
- prompt metadata and structured response schema metadata remain in LLM interaction diagnostics payloads and are not duplicated into every ledger row;
- reports reference artifact paths for utilization and ledger files through diagnostics metadata.
Redaction and retention:
- secret redaction guarantees continue to apply to diagnostics/report artifacts;
- utilization and ledger artifacts are emitted within the existing run-directory retention model (`auto|always|never`) and are retained/removed with the run directory.
Current report schema metadata values:
- `report_metadata.report_schema_name = "audita-process-report"`
- `report_metadata.report_schema_version = "v1"`
Retention modes implemented in `ApplyRetention`:
- `always`: keep all run directories.
- `never`: keep successful run directories.
- `auto`: keep failed runs and successful runs with skipped corrections.
- failed runs are always retained.
Current runtime note:
- default non-explicit runs usually have no module-level skipped corrections, so `auto` commonly removes clean successful run directories.
- explicit grammar/glossary/homophones/spoken_word runs can produce validator rejections and application skips, which are reflected in reports and retention input.
## Current tests and quality posture
Implemented tests currently cover:
- CLI argument handling and behavior (`internal/cli/run_test.go`)
- subprocess stdout/stderr and exit-code behavior (`cmd/audita/main_integration_test.go`)
- config/env/override validation (`internal/core/config/*_test.go`)
- transcript and glossary schema validation (`internal/core/schema/*_test.go`)
- deterministic normalization (`internal/core/normalization/*_test.go`)
- deterministic chunking and summaries (`internal/core/chunking/*_test.go`)
- proposal preview/apply semantics (`internal/framework/proposals/*_test.go`)
- contracts/foundation composition tests (`internal/framework/contracts/*_test.go`)
- runner sequencing and failure behavior with deterministic fake modules (`internal/framework/runner/*_test.go`)
- CLI runner integration through injected fake module factories (`internal/cli/run_test.go`)
- validator models, cardinality enforcement, and deterministic validators (`internal/framework/validators/*_test.go`)
- LLM-backed validator batching, prompt builders, structured-response safety, scheduler hooks, and diagnostics redaction (`internal/framework/validators/*_test.go`, `internal/framework/runner/*_test.go`)
- shared proposal-generation request/response parsing, deterministic indexing, scheduler hooks, and diagnostics redaction (`internal/framework/proposal_generation/*_test.go`, `internal/framework/runner/*_test.go`)
- production module-registry known-key recognition and unsupported/internal-registry error behavior (`internal/framework/modules/*_test.go`, `internal/cli/run_test.go`)
- production grammar module prompt constraints, proposal mapping, validator-chain behavior, confidence-threshold enforcement, diagnostics redaction, and explicit CLI/runtime integration (`internal/modules/grammar/*_test.go`, `internal/cli/run_test.go`, `internal/framework/runner/*_test.go`)
- production glossary module prompt constraints, proposal mapping, validator-chain behavior, confidence-threshold enforcement, diagnostics redaction, repeated-stage behavior, and explicit CLI/runtime integration (`internal/modules/glossary/*_test.go`, `internal/cli/run_test.go`, `internal/framework/runner/*_test.go`)
- production homophones module prompt constraints, proposal mapping, validator-chain behavior, confidence-threshold enforcement, diagnostics redaction, protected-term behavior, and explicit CLI/runtime integration (`internal/modules/homophones/*_test.go`, `internal/cli/run_test.go`, `internal/framework/runner/*_test.go`)
- production spoken_word module prompt constraints, proposal mapping, validator-chain behavior, semantic guardrail behavior, confidence-threshold enforcement, diagnostics redaction, protected-term behavior, and explicit CLI/runtime integration (`internal/modules/spoken_word/*_test.go`, `internal/cli/run_test.go`, `internal/framework/runner/*_test.go`)
- glossary-derived protected-term extraction and stable behavior (`internal/framework/validators/protected_terms_test.go`)
- default full-pipeline runtime shape and ordering (`internal/cli/run_test.go`, `cmd/audita/main_integration_test.go`, `internal/cli/parity_test.go`)
- subprocess operational hardening behavior including large-input, failure-mode, timeout/cancellation, backend-failure, and partial-progress paths (`cmd/audita/main_integration_test.go`)
- report/diagnostics redaction and artifact-shape behavior across success and failure paths (`internal/cli/run_test.go`, `cmd/audita/main_integration_test.go`)
- curated release-fixture and idempotence-oriented readiness checks using fake structured LLM responses (`internal/cli/release_fixtures_test.go`, `internal/cli/testdata/release`)
## Operational hardening status
The runtime now includes hardened subprocess behavior for parent-process callers:
- deterministic success/failure exit codes;
- strict stdout/stderr separation suitable for machine orchestration;
- failure stderr summaries that include diagnostics location when available;
- retained failure diagnostics (`report.json`, `error.log`, and artifacts written before failure);
- deterministic timeout/cancellation behavior in tests;
- redaction coverage for API keys/secrets across reports, diagnostics artifacts, and surfaced errors.
- stable output routing behavior:
- with `--output`, stdout remains empty on success;
- without `--output`, stdout contains only transcript JSON in the selected output schema;
- `--report-json` writes report data to file only (never stdout).
Operational caller guidance is documented in [`docs/subprocess-operations.md`](docs/subprocess-operations.md).
## Final status
- Audita's default full module-sequence runtime is implemented and tested.
- Parity fixtures and operational hardening coverage are in place.
- Historical migration context is documented in [`docs/migration-from-python.md`](docs/migration-from-python.md).

View File

@@ -63,6 +63,7 @@ Ledger records are flattened review entries derived from module results and incl
- deterministic and LLM validator decision snapshots using stable validator keys. - deterministic and LLM validator decision snapshots using stable validator keys.
Validator rejection and proposal-application skip are distinct dispositions. Validator rejection and proposal-application skip are distinct dispositions.
Module warnings are reported in module results and diagnostics metadata, but do not create standalone correction-ledger rows.
## Report references ## Report references
@@ -71,6 +72,8 @@ Validator rejection and proposal-application skip are distinct dispositions.
- correction ledger artifact; - correction ledger artifact;
- existing transcript/normalization/chunking/invocation/effective-config artifacts. - existing transcript/normalization/chunking/invocation/effective-config artifacts.
Module report entries also include warning records for malformed proposal-generation payloads and malformed validator batches.
## Retention behavior ## Retention behavior
Run-directory retention follows configured policy: Run-directory retention follows configured policy:
@@ -93,6 +96,9 @@ When debugging:
- validator rejections: - validator rejections:
- inspect `correction-ledger.json` rejected entries and matching validator decisions; - inspect `correction-ledger.json` rejected entries and matching validator decisions;
- inspect validator response diagnostics payloads. - inspect validator response diagnostics payloads.
- module warnings:
- inspect module `warnings` entries in `report.json` or `--report-json`;
- follow any diagnostic artifact path on the warning to the recorded error/response payload.
- application skips: - application skips:
- inspect `correction-ledger.json` skipped entries and skip reason codes; - inspect `correction-ledger.json` skipped entries and skip reason codes;
- compare with validator decisions to distinguish validation rejection vs apply-time skip. - compare with validator decisions to distinguish validation rejection vs apply-time skip.

View File

@@ -1,31 +1,25 @@
# Audita Public Contract # Audita Public Contract
This document defines stability expectations for Audita's external process and data interfaces.
## Scope ## Scope
This document defines stability expectations for Audita's external runtime interfaces.
This contract covers: Covered interfaces:
- CLI invocation and behavior - CLI commands and major flags;
- versioned config file behavior - versioned config behavior and precedence;
- transcript/glossary input forms - transcript/glossary input forms;
- transcript output schema selection - output schema selection;
- process report schema metadata - report schema metadata;
- stable validator key identifiers in report/diagnostics records - diagnostics artifact path metadata;
- prompt metadata identifiers in diagnostics - stdout/stderr and exit-code behavior;
- diagnostics directory behavior - redaction guarantees.
- utilization diagnostics and correction-ledger artifact presence/pathing in diagnostics metadata
- stdout/stderr and exit-code behavior
- secret redaction guarantees
- compatibility and deprecation policy
## CLI stability expectations
## CLI contract
Stable commands: Stable commands:
- `audita process` - `audita process`
- `audita config validate` - `audita config validate`
- `audita config print-effective` - `audita config print-effective`
For `audita process`, stable high-value flags include: Stable high-value `process` flags:
- `--config` - `--config`
- `--glossary` - `--glossary`
- `--output` - `--output`
@@ -33,133 +27,95 @@ For `audita process`, stable high-value flags include:
- `--modules` - `--modules`
- `--output-schema` - `--output-schema`
Compatibility flags and lower-level tuning flags remain available; they may be narrowed over time with explicit compatibility notes. ## Config contract
Supported config format:
- YAML;
- `version: 1`;
- strict unknown-field rejection.
## Config file stability expectations Path resolution for `process` and `config print-effective`:
1. `--config`
2. `AUDITA_CONFIG`
3. `/usr/local/etc/audita/config.yml`
4. `/etc/audita/config.yml`
Supported file format: Missing explicit path is an error. Missing default paths is non-fatal.
- YAML
- strict unknown-field rejection
- explicit `version`
Supported version: Precedence for `process`:
- `version: 1` 1. defaults
Precedence for `audita process`:
1. built-in defaults
2. file config 2. file config
3. environment overrides 3. environment overrides
4. CLI overrides 4. CLI overrides
Config source behavior: `config validate` remains file-only validation (defaults + file config; no env overrides).
- `--config <path>`: missing path is a clear failure
- `AUDITA_CONFIG`: missing path is a clear failure
- defaults `/usr/local/etc/audita/config.yml`, then `/etc/audita/config.yml`: both missing is non-fatal
## Supported transcript input forms Module and output-schema keys are validated against built-in catalogs. Unknown keys fail validation.
Audita accepts transcript JSON as either: ## Input contract
- a top-level array of segments Supported transcript JSON top-level forms:
- an object with a `segments` array - array of segments
- object with `segments` array
Segments must satisfy the schema and validation rules enforced by `internal/core/schema`. Supported glossary YAML form:
- top-level `glossary` list with required entry fields validated by schema parsing.
## Supported glossary input form ## Output schema contract
Supported transcript output schemas:
Audita accepts glossary YAML with a top-level `glossary` entry list and validates required fields per entry.
## Supported output schema names
Built-in output schema registry supports:
- `bare-segments` (default) - `bare-segments` (default)
- `audita-v1` - `audita-v1`
`seriatim-intermediate` is planned but not implemented. Unknown schema keys fail before output write.
Unknown output schema names fail clearly. ## Report metadata contract
Process reports include stable report metadata fields:
## Report schema/versioning expectations
Process report payloads include `report_metadata` with:
- `report_schema_name` - `report_schema_name`
- `report_schema_version` - `report_schema_version`
- `output_schema` - `output_schema`
- `config_version` when file config is used - `config_version` (when file config is loaded)
Current values: Current values:
- `report_schema_name`: `audita-process-report` - `report_schema_name = audita-process-report`
- `report_schema_version`: `v1` - `report_schema_version = v1`
`--report-json` output and diagnostics run-dir `report.json` use the same report schema metadata. `--report-json` output and run-directory `report.json` use the same report schema metadata.
Validator decision/rejection records in reports use stable validator keys in `validator_name`. Validator decision/rejection records use stable validator keys via `validator_name`.
Report diagnostics metadata includes artifact-path fields for utilization diagnostics and correction ledger when diagnostics initialization succeeds.
## Diagnostics directory behavior ## Diagnostics metadata contract
When run-directory initialization succeeds, diagnostics metadata paths reference stable artifacts, including:
- transcript and normalization artifacts;
- chunking summary;
- invocation metadata;
- redacted effective config;
- utilization diagnostics;
- correction ledger;
- `error.log` on failures.
When diagnostics directory creation succeeds, Audita writes run artifacts including: LLM interaction diagnostics include stable prompt and structured-schema identifiers where applicable.
- invocation metadata
- redacted effective config
- transcript/normalization/chunking artifacts
- utilization diagnostics (`utilization-diagnostics.json`)
- correction ledger (`correction-ledger.json`)
- report and failure error log (when applicable)
- module/LLM diagnostics artifacts as available
Retention behavior is controlled by configured retention mode; failed runs are retained. ## Stdout/stderr and exit codes
Success:
- with `--output`, stdout is empty;
- without `--output`, stdout contains transcript JSON only;
- report JSON is not written to stdout.
Diagnostics metadata for LLM interactions may include semi-public prompt identifiers: Failures:
- `prompt_id` - nonzero exit;
- `prompt_version` - human-readable stderr summary;
- `prompt_source` - diagnostics directory path on stderr when available.
- `embedded_path`
- `sha256`
These are diagnostic identifiers, not user-facing prompt override controls. Exit codes:
- `0` success
- nonzero failure
## Stdout/stderr behavior ## Redaction contract
Configured secrets are redacted from:
- effective config outputs;
- diagnostics artifacts;
- report artifacts;
- surfaced adapter/runtime errors.
Success behavior: ## Compatibility policy
- with `--output`, stdout is empty Stable command behavior, schema names, report metadata keys, diagnostics-path field semantics, and validator key identities are treated as public contract.
- without `--output`, stdout contains only transcript JSON in selected output schema
- report JSON is not written to stdout
Failure behavior: Additive fields are acceptable when existing fields and behavior remain compatible.
- stderr contains human-readable error summary
- nonzero exit
- diagnostics path is printed when available
## Exit-code behavior
- `0`: success
- nonzero: failure
Treat any nonzero exit as a failed invocation.
## Secret redaction guarantees
Audita redacts API keys and authorization secrets from:
- effective config outputs (`audita config print-effective`, diagnostics effective-config artifact)
- report artifacts
- LLM diagnostics artifacts
- surfaced request/response error messages
Config files should reference secrets via environment variable names (`api_key_env`) rather than embedding secret values.
## Compatibility and deprecation policy
- Existing stable schema names, report metadata keys, and top-level command behavior are treated as public contract.
- Compatibility inputs (legacy flags/env aliases) may remain during transition windows.
- Any planned removal or behavior change should include clear compatibility notes and migration guidance.
## Breaking changes after 1.0
After 1.0, breaking changes include, for example:
- changing default success/failure exit-code semantics
- changing stdout/stderr routing semantics
- silently changing default output schema shape
- removing supported output schema names without compatibility strategy
- changing report schema fields or meanings incompatibly
- changing config version semantics incompatibly without version bump
Additive fields, additive diagnostics, and new optional schema names are generally non-breaking when existing behavior remains intact.

View File

@@ -1,90 +1,72 @@
# Structured LLM Architecture # Structured LLM Architecture
## Purpose ## Scope
This document describes Audita's structured LLM runtime boundary and adapter behavior. This document describes Audita's structured LLM runtime boundary and adapter behavior.
## Why Audita owns the adapter ## Runtime boundary
Production LLM integration depends on the internal contract only:
Audita owns a small structured LLM adapter so that core runtime behavior is controlled inside the repository: - `contracts.StructuredLLMClient`
- request construction and schema handling are explicit and testable;
- retries, timeouts, cancellation, and error redaction are consistent across modules and validators;
- provider SDK types are not exposed outside the adapter boundary;
- dependency weight and transitive provider-specific behavior are reduced.
At runtime, the rest of Audita depends only on the internal contract:
- `StructuredLLMClient`
- `CompleteStructured(ctx, req, out)` - `CompleteStructured(ctx, req, out)`
## OpenAI-compatible request shape Provider SDK types do not leak past this boundary.
At a conceptual level, Audita sends chat completion requests with: ## Adapter ownership
- `model` `internal/framework/llm` owns the OpenAI-compatible HTTP adapter and shared LLM runtime utilities.
- `messages` (role/content pairs)
- `response_format`:
- `type = "json_schema"`
- `json_schema.name` (stable schema name)
- `json_schema.strict = true`
- `json_schema.schema` (registered JSON Schema payload)
The adapter uses OpenAI-compatible `POST {base_url}/chat/completions` over `net/http`. Key responsibilities:
- request assembly;
- timeout/cancellation propagation;
- bounded retry behavior;
- scheduler integration;
- provider response decoding;
- error redaction.
## Structured response schema registry ## Structured schema registry
Structured response schemas are registered in `internal/framework/responseschema` and include stable metadata:
- `id`
- `version`
- `name`
- `json_schema`
- `sha256`
Structured response schemas are registered in `internal/framework/responseschema` with stable metadata: Current schema keys:
- schema key - `correction_set`
- schema ID - `validator_decision_set`
- schema version
- schema name (OpenAI-compatible `response_format` name)
- raw JSON Schema payload
- SHA-256 hash
Current schemas: Schema metadata is attached to diagnostics through `Schema.DiagnosticsMap()`.
- `correction_set`:
- id `audita.correction_set`
- version `v1`
- name `audita_correction_set_v1`
- `validator_decision_set`:
- id `audita.validator_decision_set`
- version `v1`
- name `audita_validator_decision_set_v1`
## Provider compatibility assumptions ## Request shape assumptions
Audita targets OpenAI-compatible chat-completions endpoints and sends structured requests with:
- model;
- chat messages;
- `response_format.type = json_schema`;
- schema name and JSON schema payload.
Audita assumes an OpenAI-compatible chat-completions endpoint that: ## Local validation remains mandatory
- accepts message arrays with model selection; Provider schema enforcement is treated as transport-level guardrails.
- accepts `response_format.type = json_schema`;
- returns a completion with assistant message content and optional usage metadata.
Provider-specific differences are expected in strictness and error payload shapes, so the adapter treats provider output as untrusted until locally decoded. Audita still validates output locally before applying behavior changes:
- proposal decoding and proposal invariants;
- validator decision decoding and cardinality checks;
- deterministic validation and apply-time rules.
## Local decode and validation remain mandatory ## Shared malformed-output policy
Malformed structured-output classification is centralized in `internal/framework/structuredoutput`.
Provider-level structured output is a transport guardrail, not final validation. Proposal generation and validator execution both use this shared classifier so downgrade behavior cannot drift between the two paths.
After receiving a response, Audita still: ## Secrets and redaction
- decodes assistant content into typed request-specific structs; Secret extraction for LLM redaction is centralized in `llm.ConfiguredSecrets(cfg)` and reused by proposal and validator diagnostics writers.
- validates proposal and validator payload invariants locally;
- enforces deterministic validator/cardinality rules before any transcript application.
This protects runtime correctness even when provider responses are malformed, partial, or semantically inconsistent. Secrets are redacted from:
- diagnostics artifacts;
- report artifacts;
- surfaced adapter/runtime errors.
## Diagnostics and redaction ## Concurrency and scheduling
LLM execution is constrained by composed scheduler limits:
- total LLM concurrency;
- proposal LLM concurrency;
- validation LLM concurrency.
When structured schemas are used, diagnostics metadata records: The scheduler is FIFO and context-aware so permits are released on success, failure, and cancellation.
- schema ID
- schema version
- schema name
- schema hash
Diagnostics and surfaced errors preserve secret redaction:
- API keys and bearer tokens are redacted from request/response/error artifacts;
- redaction is applied before diagnostic files are written.
## Runtime behavior guarantees
The structured LLM path preserves existing runtime guarantees:
- bounded LLM call execution through schedulers;
- context-aware cancellation and timeout propagation;
- retry behavior for transient failures and retryable malformed structured responses;
- deterministic module/chunk/proposal/validator behavior outside provider nondeterminism.

View File

@@ -1,148 +1,96 @@
# Audita Validators # Audita Validators
This document describes Audita's built-in validator registry and module validator chains.
For LLM-backed validator prompt asset details, see [`docs/prompts.md`](prompts.md).
## Package ownership
Built-in validator construction is package-owned under `internal/validators/<validator_key>`:
- `internal/validators/confidence_threshold`
- `internal/validators/original_text_presence`
- `internal/validators/non_empty_corrected_text`
- `internal/validators/no_effect`
- `internal/validators/protected_terms`
- `internal/validators/spoken_form_plausibility`
- `internal/validators/meaning_reversal_review`
- `internal/validators/editorial_review`
Registry and chain wiring stay in:
- `internal/validators/registry.go`
- `internal/validators/chains.go`
Shared validator runtime mechanics stay in `internal/framework/validators`:
- request/result/decision models
- decision cardinality helpers
- protected vocabulary helpers
- shared LLM validator runtime, batching, and diagnostics helpers
Execution classification metadata is defined in `internal/validators/metadata`:
- `deterministic`
- `llm_backed`
Runner ordering uses this metadata so deterministic validators run before LLM-backed validators without concrete framework type assertions.
## Scope ## Scope
This document defines the built-in validator system used by production module runs.
Validator chains are built-in runtime behavior. ## Ownership boundaries
Built-in validator keys, constructors, and module chains are owned by `internal/validators`.
Current 1.0 boundary: Shared runtime execution mechanics are owned by `internal/framework/validators`, including:
- built-in validator keys and built-in module chains are stable runtime identifiers; - validator request/result models;
- thresholds and batching knobs remain configurable where already supported; - deterministic proposal checks;
- arbitrary user-defined validator chains are deferred. - LLM validator batching and execution;
- decision-cardinality enforcement;
- diagnostics integration.
## Built-in validator keys Execution class metadata is owned by `internal/validators/metadata`.
### Deterministic validators
## Stable validator keys
Deterministic:
- `proposal_shape`
- `confidence_threshold` - `confidence_threshold`
- checks proposal confidence against module-specific configured threshold.
- `original_text_presence` - `original_text_presence`
- ensures target segment exists and `original_text` exists in current working segment text.
- `non_empty_corrected_text` - `non_empty_corrected_text`
- rejects blank/whitespace-only `corrected_text`.
- `no_effect` - `no_effect`
- rejects proposals where `original_text == corrected_text`.
- `protected_terms` - `protected_terms`
- protects glossary-derived terms from unsafe mutations in non-glossary modules.
- glossary stages use glossary-specific protection logic but still report this same stable key.
### LLM-backed validators
LLM-backed:
- `spoken_form_plausibility` - `spoken_form_plausibility`
- checks whether proposed spoken-form change remains plausible in transcript context.
- `meaning_reversal_review` - `meaning_reversal_review`
- checks for likely meaning reversal or semantic contradiction.
- `editorial_review` - `editorial_review`
- performs conservative editorial safety review.
## Built-in module chains ## Built-in module chains
`glossary`:
- `proposal_shape`
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `spoken_form_plausibility`
- `meaning_reversal_review`
Current built-in chains resolved from `internal/validators/chains.go`: `homophones`:
- `proposal_shape`
- `no_effect`
- `original_text_presence`
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `spoken_form_plausibility`
- `meaning_reversal_review`
- `glossary` `spoken_word`:
- `no_effect` - `proposal_shape`
- `original_text_presence` - `no_effect`
- `confidence_threshold` - `original_text_presence`
- `protected_terms` - `confidence_threshold`
- `non_empty_corrected_text` - `protected_terms`
- `spoken_form_plausibility` - `non_empty_corrected_text`
- `meaning_reversal_review` - `editorial_review`
- `meaning_reversal_review`
- `homophones` `grammar`:
- `no_effect` - `proposal_shape`
- `original_text_presence` - `no_effect`
- `confidence_threshold` - `original_text_presence`
- `protected_terms` - `confidence_threshold`
- `non_empty_corrected_text` - `protected_terms`
- `spoken_form_plausibility` - `non_empty_corrected_text`
- `meaning_reversal_review` - `editorial_review`
- `meaning_reversal_review`
- `spoken_word` ## Ordering and execution semantics
- `no_effect` Validator ordering is based on canonical metadata:
- `original_text_presence` - deterministic validators run before LLM-backed validators.
- `confidence_threshold`
- `protected_terms`
- `non_empty_corrected_text`
- `editorial_review`
- `meaning_reversal_review`
- `grammar` Within each module stage:
- `no_effect` - proposals are generated per section;
- `original_text_presence` - validator chains execute on those proposals;
- `confidence_threshold` - approved proposals are applied once after section work settles.
- `protected_terms`
- `non_empty_corrected_text`
- `editorial_review`
- `meaning_reversal_review`
## Protected terms construction ## Malformed payload behavior
Malformed structured-output from proposal generation and LLM validator calls is downgraded, not treated as a process-fatal transport error.
`protected_terms` has explicit constructors: Current outcomes:
- general constructor used by non-glossary modules through the built-in registry - malformed proposal-generation payloads produce section/module warnings and zero proposals for the affected section;
- glossary-stage constructor used by glossary chain resolution - malformed validator decision payloads reject the affected validator batch with warnings;
- deterministic validator behavior and runner order remain unchanged.
Both variants preserve existing behavior and report the stable key `protected_terms`. ## Reporting identity
Reports and diagnostics use stable validator keys as identifiers.
## Execution semantics Correction-ledger deterministic-vs-LLM classification is derived from canonical validator metadata, not package-local hardcoded maps.
- modules execute serially; ## Prompt assets
- section proposal work can run concurrently within a module; LLM validator prompt assets and prompt metadata are documented in [Prompts](./prompts.md).
- deterministic validators run before LLM-backed validators;
- malformed/missing/duplicate/unknown LLM validator decisions fail safely;
- approved proposals are applied once per module after section work settles.
## Validator rejections vs proposal-application skips
- validator rejection:
- proposal is denied by validator-chain review and appears in validator rejection reporting with validator key and reason code.
- proposal-application skip:
- proposal passed validators but could not be applied under replacement-policy semantics (for example no matching span at apply time).
These are separate outcomes and are reported separately.
## Reporting and diagnostics identity
- report validator decision/rejection entries use stable validator keys in `validator_name`.
- validator LLM diagnostics include validator identity in interaction metadata and structured response schema metadata.
- correction ledger entries include deterministic and LLM validator decision snapshots keyed by the same stable validator keys, and keep validator rejection distinct from application-level skip.
Prompt assets are unchanged by the validator package-ownership refactor and remain built-in under `internal/prompts`.
## Configurable knobs that remain supported
- per-module confidence thresholds (`thresholds.*` / equivalent env+CLI overrides)
- validation batching limits (`validation_max_prompt_tokens` / equivalent env+CLI overrides)
- validation LLM model/base URL/timeout/retries/concurrency settings
These tune validator behavior without exposing arbitrary user-defined chains.

View File

@@ -1,51 +1,48 @@
# Audita Configuration # Audita Configuration
This document describes Audita's versioned YAML config support and related commands. ## Scope
This document defines the supported versioned YAML configuration model and runtime precedence behavior.
## Purpose
Audita's config file provides a stable place for pipeline defaults and runtime tuning that would otherwise require many environment variables or CLI flags.
Use config files for baseline settings, then use environment variables and CLI flags for deployment and per-run overrides.
## Supported version
Current supported config version:
## Supported file version
Current supported config file version:
- `version: 1` - `version: 1`
Rules: Validation rules:
- missing `version` fails;
- missing `version` fails validation; - unsupported version fails;
- unknown versions fail validation; - unknown YAML fields fail (strict decoding).
- unknown fields fail validation (strict decoding).
## Config path resolution ## Config path resolution
For `audita process` and `audita config print-effective`, path resolution order is:
1. `--config <path>`
2. `AUDITA_CONFIG`
3. `/usr/local/etc/audita/config.yml` (if present)
4. `/etc/audita/config.yml` (if present)
For `audita process`, config path resolution is: Missing-path behavior:
- missing `--config` path is an error;
- missing `AUDITA_CONFIG` path is an error;
- missing both default paths is non-fatal.
1. `--config <path>` if provided ## Effective precedence
2. `AUDITA_CONFIG` if set and `--config` is not provided `audita process` effective precedence:
3. default `/usr/local/etc/audita/config.yml` if present 1. defaults
4. fallback default `/etc/audita/config.yml` if present
Missing-file behavior:
- missing `--config` path: hard failure;
- missing `AUDITA_CONFIG` path: hard failure;
- missing both default-path files: non-fatal, run continues.
## Precedence model
Effective config precedence is:
1. built-in defaults
2. file config 2. file config
3. environment overrides 3. environment overrides
4. CLI overrides 4. CLI overrides
## Supported YAML fields `audita config print-effective` uses:
1. defaults
2. file config
3. environment overrides
`audita config validate` intentionally uses file-only validation:
1. defaults
2. file config
Environment overrides are not applied in `config validate`.
## Supported top-level YAML fields
```yaml ```yaml
version: 1 version: 1
@@ -62,7 +59,6 @@ llm:
api_key_env: AUDITA_LLM_API_KEY api_key_env: AUDITA_LLM_API_KEY
timeout: 120s timeout: 120s
max_retries: 3 max_retries: 3
validation: validation:
base_url: https://openrouter.ai/api/v1 base_url: https://openrouter.ai/api/v1
model: openrouter/google/gemma-4-31b-it model: openrouter/google/gemma-4-31b-it
@@ -100,91 +96,54 @@ diagnostics:
retention: auto retention: auto
``` ```
`context.description` provides background-only transcript context for prompts. ## Module and output-schema validation
If both config and CLI provide a description, `--transcript-description` takes precedence. `pipeline.modules` keys are validated against the built-in supported module catalog.
`output.schema` supports the built-in output schema registry values: Supported module keys:
- `bare-segments` (default) - `glossary`
- `homophones`
- `spoken_word`
- `grammar`
Repeated supported module keys are allowed.
`output.schema` is validated against the built-in output schema catalog.
Supported output schema keys:
- `bare-segments`
- `audita-v1` - `audita-v1`
Unknown schema names fail clearly before transcript output is written. Unknown module keys and unknown output schema keys fail validation.
Duration-like fields accept either: ## Duration field parsing
Duration-like fields support:
- numeric seconds (for example `120`, `3.5`)
- duration strings (for example `120s`, `2m`)
- numeric seconds (for example `120`, `3.5`), or LLM timeout duration strings must resolve to whole seconds.
- duration strings (for example `120s`, `2m`).
For LLM timeouts, duration strings must resolve to whole seconds.
## Secret handling ## Secret handling
Use `api_key_env` fields for secrets:
Use `api_key_env` for secrets:
- `llm.proposal.api_key_env` - `llm.proposal.api_key_env`
- `llm.validation.api_key_env` - `llm.validation.api_key_env`
These fields must contain environment variable names, not secret values. These fields store environment variable names, not secret values.
At runtime, Audita resolves those names from the process environment. Resolved secret values are redacted from:
- `audita config print-effective` output;
Redaction behavior: - diagnostics `effective-config.json`;
- report and diagnostics payloads.
- run diagnostics `effective-config.json` is redacted;
- `audita config print-effective` output is redacted;
- API keys are never emitted in plaintext by those outputs.
## Config commands
Validate a config file:
## Commands
Validate a file config:
```sh ```sh
audita config validate --config ./audita.yml audita config validate --config ./audita.yml
``` ```
Print redacted effective config: Print redacted effective config:
```sh ```sh
audita config print-effective --config ./audita.yml audita config print-effective --config ./audita.yml
``` ```
`print-effective` loads defaults, then file config, then environment overrides.
## Example: local OpenAI-compatible endpoint
```yaml
version: 1
llm:
proposal:
base_url: http://localhost:8000/v1
model: local/proposal-model
api_key_env: AUDITA_LLM_API_KEY
timeout: 90s
max_retries: 2
validation:
base_url: http://localhost:8000/v1
model: local/validation-model
api_key_env: AUDITA_VALIDATION_LLM_API_KEY
timeout: 90s
max_retries: 2
pipeline:
modules: [glossary, homophones, glossary, spoken_word, grammar]
diagnostics:
work_dir: /tmp/audita
retention: auto
```
## Compatibility notes ## Compatibility notes
Legacy compatibility flags and environment aliases remain available where implemented, but the stable configuration surface is the versioned YAML model described above.
Existing environment variables and lower-level CLI flags remain available for compatibility.
Current guidance:
- prefer file config for baseline behavior;
- keep environment variables for secrets/deployment-specific overrides;
- use CLI flags for per-run overrides.
- validator chains are built-in and are not user-configurable in config.
- prompt source selection and filesystem prompt overrides are not config options.

33
docs/development.md Normal file
View File

@@ -0,0 +1,33 @@
# Audita Development Workflow
## Scope
This document defines the canonical contributor workflow and engineering conventions for this repository.
## Workflow
1. Start from a clean understanding of scope and constraints.
2. Make focused changes that preserve existing public behavior unless behavior change is explicitly intended.
3. Run targeted tests for touched packages.
4. Run `go test ./...` before finalizing substantial changes.
5. Update affected documentation so it describes current behavior only.
## Engineering conventions
- Keep module packages separate: `glossary`, `homophones`, `spoken_word`, `grammar`.
- Prefer narrow shared helpers and catalogs over broad abstractions.
- Preserve diagnostics artifact naming and report field contracts unless intentionally changed.
- Preserve CLI/config precedence semantics unless intentionally changed.
- Treat stable validator keys, prompt identifiers, and output-schema keys as contract surfaces.
## Configuration and runtime expectations
- `audita process` precedence is defaults -> file -> env -> CLI.
- `audita config validate` validates file config merged onto defaults only.
- `audita config print-effective` includes environment overrides and prints redacted JSON.
## Testing expectations
- Add tests for new behavior and for bug fixes.
- Keep deterministic fixtures stable.
- Do not reduce existing parity, release-fixture, subprocess, or module-specific coverage without equivalent replacement.
## Commit discipline
- Keep commits scoped and reviewable.
- Avoid mixing unrelated refactors with behavior changes.
- Use clear plain-English commit messages.

View File

@@ -0,0 +1,27 @@
# Documentation Policy
## Scope
This policy defines how project documentation should be authored and maintained.
## Core rules
- Document the current behavior of the codebase.
- Remove stale behavior descriptions promptly when code changes.
- Do not describe development history in architecture or behavior docs unless a document is explicitly historical.
- Do not use architecture or behavior docs as changelogs.
- Prefer rewriting stale sections from scratch when substantial behavior or ownership changes occur.
## Consistency requirements
- Keep command examples aligned with current CLI surfaces.
- Keep configuration examples aligned with supported fields and precedence.
- Keep architecture package ownership descriptions aligned with current code layout.
- Keep stable contract identifiers accurate (module keys, validator keys, output-schema keys, report metadata fields).
## Cross-document expectations
- `docs/architecture/*` documents runtime behavior and package ownership.
- `docs/configuration.md` documents config schema and precedence.
- `docs/development.md` documents contributor workflow and engineering conventions.
## Review expectations for documentation changes
- Verify referenced files and links exist.
- Verify examples match current behavior.
- Prefer concise, direct language and avoid speculative future claims.

View File

@@ -40,6 +40,7 @@ Use this checklist before cutting a pre-1.0 or 1.0 release candidate.
- Verify structured response schemas are attached via `response_format.type=json_schema`. - Verify structured response schemas are attached via `response_format.type=json_schema`.
- Verify diagnostics metadata includes structured schema `id/version/name/sha256`. - Verify diagnostics metadata includes structured schema `id/version/name/sha256`.
- Verify provider output is still locally decoded/validated before use. - Verify provider output is still locally decoded/validated before use.
- Verify malformed module-stage structured payloads degrade to warnings/rejections instead of failing the run.
## Report and diagnostics schema checks ## Report and diagnostics schema checks
@@ -67,6 +68,7 @@ Use this checklist before cutting a pre-1.0 or 1.0 release candidate.
- Verify prompt metadata appears in LLM request metadata diagnostics: - Verify prompt metadata appears in LLM request metadata diagnostics:
- `prompt_id`, `prompt_version`, `prompt_source`, `embedded_path`, `sha256`. - `prompt_id`, `prompt_version`, `prompt_source`, `embedded_path`, `sha256`.
- Verify stable validator keys appear in report decisions/rejections. - Verify stable validator keys appear in report decisions/rejections.
- Verify module warning records appear in reports for malformed proposal-generation payloads and malformed validator batches.
- Verify built-in validator chains resolve and execute for default and explicit module runs. - Verify built-in validator chains resolve and execute for default and explicit module runs.
## Utilization diagnostics checks ## Utilization diagnostics checks
@@ -96,6 +98,8 @@ Use this checklist before cutting a pre-1.0 or 1.0 release candidate.
## Failure and cancellation checks ## Failure and cancellation checks
- Verify controlled failure paths retain diagnostics and produce best-effort failure reports. - Verify controlled failure paths retain diagnostics and produce best-effort failure reports.
- Verify malformed proposal-generation payloads keep exit code `0`, keep stderr empty on success, and record warnings in reports/diagnostics.
- Verify malformed validator payloads reject only the affected batch and do not fail the module.
- Verify timeout/cancellation paths exit nonzero, do not hang, and retain failure diagnostics when initialized. - Verify timeout/cancellation paths exit nonzero, do not hang, and retain failure diagnostics when initialized.
## Release fixture/idempotence checks ## Release fixture/idempotence checks

791
docs/roadmap/audit.md Normal file
View File

@@ -0,0 +1,791 @@
# Pre-1.0 Code Quality and Deduplication Audit
## 1. Executive summary
Audita is in good shape for a limited pre-1.0 cleanup pass. The repository is small, package boundaries are mostly explicit, and the core public contract is already documented around `audita process`, config loading, output schemas, diagnostics, reports, embedded prompts, modules, and validators. The highest-value improvements are targeted centralization, not a rewrite.
Top three refactoring targets before 1.0:
1. Centralize module proposal plumbing and prompt payload construction across the four production modules.
2. Centralize effective config loading plus schema/module catalog validation so `process`, `config print-effective`, and `config validate` cannot drift.
3. Centralize diagnostics artifact names, stage names, and validator classification metadata used by reports and the correction ledger.
No major architectural risk is apparent. The main pre-1.0 risk is public-behavior drift from repeated policy strings, catalog values, artifact paths, and nearly identical command/module scaffolding.
This report was written to `docs/roadmap/audit.md`. `docs/roadmap/` already exists in the repository, although its previous `publish.md` file is currently deleted in the worktree by an unrelated change.
## 2. Repository map reviewed
Reviewed documentation:
- `README.md`
- `docs/configuration.md`
- `docs/architecture/architecture.md`
- `docs/architecture/public-contract.md`
- `docs/architecture/diagnostics.md`
- `docs/architecture/output-schemas.md`
- `docs/architecture/prompts.md`
- `docs/architecture/validators.md`
- `docs/architecture/structured-llm.md`
- `docs/integration/subprocess-operations.md`
- `docs/release-checklist.md`
Reviewed implementation areas:
- `cmd/audita`
- `internal/cli`
- `internal/core/config`
- `internal/core/schema`
- `internal/core/io`
- `internal/core/normalization`
- `internal/core/chunking`
- `internal/core/diagnostics`
- `internal/core/outputschema`
- `internal/core/reporting`
- `internal/framework/contracts`
- `internal/framework/modules`
- `internal/framework/proposal_generation`
- `internal/framework/proposals`
- `internal/framework/runner`
- `internal/framework/validators`
- `internal/framework/llm`
- `internal/framework/responseschema`
- `internal/framework/promptcontext`
- `internal/framework/warnings`
- `internal/modules/glossary`
- `internal/modules/homophones`
- `internal/modules/spoken_word`
- `internal/modules/grammar`
- `internal/prompts`
- `internal/validators`
- package tests and CLI parity/release fixtures under `internal/cli/testdata`
Major execution paths reviewed:
- `audita process <transcript.json> --glossary <glossary.yaml>`
- `audita config validate --config <path>`
- `audita config print-effective [--config <path>]`
- default module sequence resolution and repeated glossary instance naming
- proposal generation, validator execution, proposal application, report writing, diagnostics writing, and retention
Important absent or not-applicable areas:
- No `pkg/` directory exists.
- No `examples/` directory exists.
- No `docs/internal/` directory exists.
- No `internal/app`, `internal/stage`, `internal/storage`, `internal/artifacts`, or `internal/manifest` packages exist. Their closest equivalents are `internal/cli`, `internal/framework/runner`, `internal/core/diagnostics`, and `internal/core/reporting`.
## 3. High-confidence deduplication opportunities
### 3.1 Module proposal plumbing is duplicated across all production modules
Affected files/packages:
- `internal/modules/glossary/module.go`
- `internal/modules/homophones/module.go`
- `internal/modules/spoken_word/module.go`
- `internal/modules/grammar/module.go`
- `internal/modules/*/prompt.go`
- `internal/framework/proposal_generation`
- `internal/framework/promptcontext`
Duplicated or near-duplicated behavior:
- Each module has the same `Module` struct shape, `Validators` copy behavior, `Propose` flow, section transcript extraction, transcript description extraction, `proposal_generation.GenerateCandidates` request construction, prompt metadata map construction, and stage-name formatting.
- Each module also has a near-identical prompt payload builder with local `promptSegment` and `promptTranscriptSection` types, glossary JSON marshaling, transcript section JSON marshaling, transcript description block rendering, and two-message return shape.
- `collectSectionProposals` already passes a section transcript to each module, but each module then filters that transcript again by section metadata.
Why it matters:
- A diagnostics or prompt-context bug fix would need to be repeated in four modules.
- Prompt metadata fields and stage names are diagnostics-visible and could drift by module.
- The double section filtering is currently harmless, but it obscures the runner/module contract.
Recommended refactor:
- Add a small shared helper for module proposal execution, likely in `internal/framework/proposal_generation` or a narrow `internal/modules/modulekit` package.
- Keep domain-specific prompt IDs and prompt text local to each module.
- Move transcript section prompt payload construction into a shared prompt-context helper, for example `promptcontext.MarshalTranscriptSection`.
- Provide one helper for prompt metadata maps instead of manually expanding `prompt_id`, `prompt_version`, `prompt_source`, `embedded_path`, and `sha256` in every module.
- Preserve current module `Key`, replacement policy, and validator chain ownership.
Suggested tests:
- Keep one golden or table-driven prompt payload test per module for domain-specific wording.
- Add shared tests for transcript section JSON shape, empty transcript handling, categories copy behavior, and prompt metadata fields.
- Add a parity test that all four module `Propose` methods still write diagnostics under the same module instance directory and produce the same correction mapping.
Risk level:
- Low to medium. The behavior is highly duplicated, but prompt and diagnostics behavior is sensitive. Refactor behind existing module tests and CLI parity fixtures.
### 3.2 Effective config loading is repeated between commands
Affected files/packages:
- `internal/cli/run.go`
- `internal/core/config`
Duplicated or near-duplicated behavior:
- `runProcess` and `runConfigPrintEffective` both resolve config path, start from defaults, optionally load/apply file config, then apply environment overrides.
- `runConfigValidate` separately loads a file, applies it to defaults, and validates it.
- Path source metadata is computed in `internal/cli`, not `internal/core/config`, even though the precedence contract is documented as config behavior.
Why it matters:
- Config precedence is part of the public contract. If a future setting is added, three command paths may need coordinated updates.
- `config print-effective` is the user-visible diagnostic for effective config. It should use the same loader as `process`, except for intentionally omitted CLI overrides.
- The current code is understandable, but the behavior is repeated in a way that makes drift likely as config grows.
Recommended refactor:
- Add a narrow effective-config loader in `internal/core/config`, returning `Config`, source path, source type, and version metadata.
- Keep command-specific CLI flag parsing in `internal/cli`.
- Model the intentional differences explicitly:
- `process`: defaults + file + env + CLI overrides
- `config print-effective`: defaults + file + env
- `config validate`: file schema + default-backed config validation, no env
- Move `resolveConfigPath` or an equivalent path resolver into `internal/core/config`.
Suggested tests:
- One table-driven config loader test covering explicit `--config`, `AUDITA_CONFIG`, default search paths, missing explicit paths, and missing default paths.
- CLI tests asserting `process` and `print-effective` share file+env behavior.
- A regression test that `config validate` remains file-only and does not read environment overrides.
Risk level:
- Low. Behavior is already explicit and well tested; the refactor can be done by moving code without changing precedence.
### 3.3 Module catalog validation is split across config, contracts, and module factory
Affected files/packages:
- `internal/core/config/validation.go`
- `internal/framework/contracts/contracts.go`
- `internal/framework/modules/registry.go`
- `internal/validators/chains.go`
- `internal/framework/validators/models.go`
Duplicated or near-duplicated behavior:
- Module keys appear in multiple places:
- config default CSV: `glossary,homophones,glossary,spoken_word,grammar`
- module factory constants and known-key map
- built-in validator chains
- confidence threshold lookup
- individual module `Key()` methods
- `Config.Validate` checks only that module names are non-empty. An unsupported configured module can pass `audita config validate` and fail later in `process` runner setup.
- `contracts.ResolveModuleRunSpecs` only assigns instance names; it does not validate production module support.
Why it matters:
- `audita config validate` is documented as a CI/preflight command. Letting unsupported modules pass weakens that preflight.
- Module key drift could affect thresholds, validator chains, reports, and unsupported-module errors.
Recommended refactor:
- Introduce a small canonical module catalog or key package that can be imported by config validation, module factory construction, validator chain resolution, and threshold lookup without creating a cycle.
- Keep module construction in `internal/framework/modules`; the catalog should expose keys and validation only.
- Make `Config.Validate` reject unknown built-in module keys through that catalog.
- Keep repeated module instances valid.
Suggested tests:
- `internal/core/config` test: unknown `pipeline.modules` fails validation.
- `internal/cli` test: `audita config validate --config` rejects an unsupported module before runtime.
- Existing `internal/framework/modules` unknown-module tests should continue to pass.
- Validator chain tests should assert every catalog module has a built-in chain.
Risk level:
- Medium. This tightens validation behavior. It is desirable before 1.0, but if unknown modules were intentionally allowed for future extension, document that explicitly instead.
### 3.4 Output schema support is hardcoded in config validation and registry
Affected files/packages:
- `internal/core/config/validation.go`
- `internal/core/outputschema/registry.go`
- `docs/architecture/output-schemas.md`
Duplicated or near-duplicated behavior:
- `Config.Validate` hardcodes `bare-segments` and `audita-v1`.
- `outputschema.Resolve` owns the actual output schema registry and returns the runtime error for unsupported schema names.
Why it matters:
- Adding or deferring a schema requires updating multiple places.
- Public behavior could drift: a schema might validate in config but fail at output time, or vice versa.
Recommended refactor:
- Make `internal/core/outputschema` expose `IsSupported`, `SupportedKeys`, or a validation function.
- Have config validation call that helper or consume shared constants.
- Keep actual encoding logic in `outputschema`; config should not know encoder details.
Suggested tests:
- Config validation test for every output schema returned by the registry.
- Output schema registry test that unsupported `seriatim-intermediate` still fails clearly until implemented.
- CLI test that unsupported `--output-schema` fails before output write.
Risk level:
- Low. This is a straightforward catalog centralization.
### 3.5 Diagnostics artifact names and report metadata paths are repeated
Affected files/packages:
- `internal/core/diagnostics/run_dir.go`
- `internal/cli/run.go`
- `internal/core/reporting/report.go`
- docs under `docs/architecture` and `docs/integration`
Duplicated or near-duplicated behavior:
- Artifact filenames such as `source-transcript.json`, `source-transcript-parsed.json`, `normalized-transcript.json`, `normalization-summary.json`, `chunking-summary.json`, `utilization-diagnostics.json`, `correction-ledger.json`, `invocation.json`, `effective-config.json`, `report.json`, and `error.log` are repeated between run-directory writers and `buildProcessReport`.
- `runProcess` writes `utilization-diagnostics.json` and `correction-ledger.json` by raw string on both success and failure paths.
Why it matters:
- These names are part of the documented diagnostics contract.
- A filename change would need to be made in multiple places, and report metadata could point at files that are no longer written.
Recommended refactor:
- Define diagnostics artifact name constants in `internal/core/diagnostics`.
- Add a helper that returns `reporting.DiagnosticsMetadata` for a run directory and status.
- Add named methods for utilization diagnostics and correction ledger writes, or at least constants used by `WriteJSONArtifact`.
Suggested tests:
- Unit test that `diagnostics.MetadataForRunDirectory` matches files written by `RunDirectory`.
- CLI success/failure tests should continue to assert report metadata paths and actual file existence.
- Add a test for failure report metadata including `error.log`.
Risk level:
- Low. This is mostly string centralization, with high public-contract value.
### 3.6 Validator execution class is duplicated and partially hardcoded
Affected files/packages:
- `internal/validators/registry.go`
- `internal/validators/metadata/metadata.go`
- `internal/validators/*/validator.go`
- `internal/framework/runner/runner.go`
- `internal/cli/review_artifacts.go`
Duplicated or near-duplicated behavior:
- Validator constructors wrap validators with execution class metadata.
- `BuiltInValidatorDefinition` also has an `LLMBacked` field.
- Runner uses `metadata.ClassOf` to order deterministic validators before LLM-backed validators.
- Correction ledger classification uses a local hardcoded map of LLM-backed validator names.
Why it matters:
- Adding a new LLM-backed validator could be ordered correctly by runner metadata but appear in the wrong correction-ledger section.
- Validator class is domain metadata, not report-building policy. It should have one source of truth.
Recommended refactor:
- Make validator classification resolvable by validator instance or stable key from a single metadata source.
- Remove the unused or redundant `LLMBacked` field, or make it the canonical source used by constructors, runner ordering, and ledger formatting.
- Replace the local ledger map with `metadata.ClassOf` when possible, or a registry lookup by stable key.
Suggested tests:
- Correction ledger test that LLM-backed decisions are classified from validator metadata, not a local string map.
- Registry test that every registered LLM-backed validator reports the same class through every public metadata path.
- Runner ordering test should remain in place.
Risk level:
- Low to medium. The implementation is small, but correction-ledger shape is diagnostics-visible.
### 3.7 Malformed structured-output classification is duplicated
Affected files/packages:
- `internal/framework/proposal_generation/generate.go`
- `internal/framework/validators/llm_validators.go`
- `internal/framework/llm/openai_compatible_client.go`
Duplicated or near-duplicated behavior:
- Proposal generation and LLM validators both classify malformed structured-output errors by scanning error message substrings.
- The marker lists are currently the same, but they are maintained independently.
- The actual errors originate in the LLM adapter.
Why it matters:
- Proposal-generation malformed payloads become warnings with zero proposals, while validator malformed payloads reject affected batches with warnings. If classifiers drift, similar adapter failures could be downgraded in one workflow and hard-fail in another.
Recommended refactor:
- Prefer a typed error or exported classifier from `internal/framework/llm`.
- If typed errors are too invasive, create one shared classifier function in a lower framework package used by both proposal generation and validators.
- Preserve the different handling semantics at each call site.
Suggested tests:
- Shared classifier table for all adapter malformed-output errors.
- Proposal-generation test and validator test should assert the same representative malformed adapter errors are downgraded.
- Adapter tests should assert typed/classified errors wrap useful context and still redact secrets.
Risk level:
- Medium. Error typing can accidentally affect retry and wrapping behavior; do this with focused tests.
## 4. Medium-confidence opportunities
### 4.1 CLI flag registration and override extraction are large and repetitive
Affected files/packages:
- `internal/cli/run.go`
- `internal/core/config/flags.go`
Duplicated or near-duplicated behavior:
- Each process flag has a field in `processFlags`, a registration entry in `newProcessFlagSet`, a case in `fs.Visit`, and an assignment in `config.ApplyCLIOverrides`.
- File config and environment config also set many of the same effective config fields.
Why it matters:
- Adding a new config option requires multiple edits. Missing one edit could create a flag that displays but does not override, or a config field with no CLI override.
Recommended refactor:
- Avoid a generic reflection-heavy flag system before 1.0.
- Consider a small metadata table only for simple scalar flags, or a focused helper that maps visited flags to `CLIOverrides`.
- Keep nontrivial semantics, such as legacy concurrency alias precedence, explicit in code.
Suggested tests:
- CLI override parity test for every stable flag that mutates config.
- A test that default flag values reflect file+env effective config before CLI overrides.
Risk level:
- Medium. A broad flag abstraction would be riskier than the current duplication. Do only a small helper if it clearly reduces missed updates.
### 4.2 Config source application repeats field-level assignments
Affected files/packages:
- `internal/core/config/file_config.go`
- `internal/core/config/env.go`
- `internal/core/config/flags.go`
Duplicated or near-duplicated behavior:
- The same effective fields are assigned from file config, env vars, and CLI overrides.
- Some semantics differ intentionally: file config supports `api_key_env`, env supports `OPENROUTER_API_KEY` fallback, CLI uses direct values.
Why it matters:
- Field additions are easy to miss in one source.
- Error messages and trimming behavior can drift.
Recommended refactor:
- Do not force all config sources through one generic mapper.
- Add small setter helpers for repeated config subdomains such as LLM target, concurrency, thresholds, normalization, and diagnostics.
- Keep source-specific parsing and error labels local.
Suggested tests:
- Cross-source table proving file, env, and CLI all reach the same effective fields where they are meant to.
- Tests for intentional differences: API key env resolution, `OPENROUTER_API_KEY` fallback, CLI direct API key, and transcript description trimming.
Risk level:
- Medium. Useful, but only after the effective loader and catalog cleanup.
### 4.3 Prompt metadata and response schema metadata map construction repeats
Affected files/packages:
- `internal/modules/*/module.go`
- `internal/framework/proposal_generation/generate.go`
- `internal/framework/validators/llm_validators.go`
- `internal/prompts`
- `internal/framework/responseschema`
Duplicated or near-duplicated behavior:
- Prompt metadata maps are manually expanded in module proposal generation and validator diagnostics.
- Response schema metadata maps are built independently in proposal generation and validator diagnostics.
Why it matters:
- Metadata fields are diagnostics-visible and useful for reproducibility.
- Adding a metadata field requires updating multiple call sites.
Recommended refactor:
- Add `Metadata.Map()` or a typed diagnostics metadata struct in `internal/prompts`.
- Add `responseschema.Metadata()` or a method returning a stable diagnostics shape.
- Prefer typed structs over `map[string]any` where possible.
Suggested tests:
- Prompt metadata rendering test should assert all registered prompts expose stable metadata.
- Proposal and validator diagnostics tests should assert the shared metadata helper is used.
Risk level:
- Low.
### 4.4 Secret redaction logic is split across config, LLM diagnostics, and adapter errors
Affected files/packages:
- `internal/core/config/redaction.go`
- `internal/framework/llm/diagnostics.go`
- `internal/framework/llm/client_common.go`
- `internal/framework/proposal_generation/generate.go`
- `internal/framework/runner/runner.go`
Duplicated or near-duplicated behavior:
- Config redaction replaces non-empty API keys with `[REDACTED]`.
- LLM diagnostics replace configured secret values and `Bearer <secret>`.
- Adapter error sanitization separately replaces secrets and bearer values.
- Proposal and validator paths separately assemble secret lists.
Why it matters:
- Secret redaction is a public guarantee.
- New secret-bearing config fields could be missed in one path.
Recommended refactor:
- Add a small redaction helper package or keep it in `internal/framework/llm` only if it remains LLM-specific.
- Centralize `[]string` secret extraction from `config.Config`.
- Keep config structural redaction separate from byte/string payload redaction, but share the redaction token and value replacement behavior.
Suggested tests:
- One test that a proposal-generation error, validator diagnostic artifact, effective config artifact, and surfaced provider error all redact the same configured secrets.
- Existing subprocess no-secret-leak test should remain as an end-to-end guard.
Risk level:
- Medium. The current coverage appears strong; change carefully.
### 4.5 Test fakes and fixture helpers are duplicated across packages
Affected files/packages:
- `internal/modules/*/module_test.go`
- `internal/framework/proposal_generation/generate_test.go`
- `internal/framework/validators/llm_validators_test.go`
- `internal/cli/run_test.go`
- `cmd/audita/main_integration_test.go`
- `internal/cli/release_fixtures_test.go`
- `internal/cli/parity_test.go`
Duplicated or near-duplicated behavior:
- Several packages define fake structured LLM clients, fixture path helpers, read/write helpers, diagnostics glob assertions, and run-directory helpers.
- The four module test files have particularly similar fake clients and proposal-diagnostics assertions.
Why it matters:
- Refactors in LLM or diagnostics behavior require updating many tests.
- Some duplicated tests are valuable because they preserve per-module public behavior; the issue is helper duplication, not coverage volume.
Recommended refactor:
- Add package-local helper files where duplication is within a package.
- For cross-package fakes, prefer a small internal test support package only if it does not create import cycles or hide test intent.
- Keep module-specific assertions local.
Suggested tests:
- This is test infrastructure cleanup. Existing tests should remain semantically equivalent.
- Add helper tests only if helpers contain nontrivial behavior, such as fake response sequencing.
Risk level:
- Low.
### 4.6 Stage-name construction is inconsistent enough to centralize, but not enough to redesign
Affected files/packages:
- `internal/modules/*/module.go`
- `internal/framework/proposal_generation/generate.go`
- `internal/framework/validators/llm_validators.go`
- `internal/framework/runner/observability.go`
Duplicated or near-duplicated behavior:
- Modules pass stage names like `<module_instance>:proposal:section-0001`.
- `proposal_generation` has a default builder using `<module_instance>:proposal-generation:section-0001`, but production modules bypass it.
- Validators build `<module_instance>:<validator>:batch-0001`.
- Utilization extracts module instance by splitting stage names on `:`.
Why it matters:
- Stage names affect diagnostics filenames and observability grouping.
- Current behavior works, but the naming grammar is implicit.
Recommended refactor:
- Add narrow helpers for proposal and validator stage names.
- Preserve current production stage names unless there is a deliberate pre-1.0 diagnostics compatibility decision.
- Keep filename sanitization in `internal/framework/llm`.
Suggested tests:
- Unit tests for stage-name helper output.
- Utilization test that module instance extraction still works for proposal and validator stage names.
Risk level:
- Medium. Renaming stages can change diagnostics filenames, so avoid unnecessary churn.
## 5. Boundary and responsibility concerns
### CLI owns too much report and diagnostics metadata assembly
`internal/cli/run.go` is doing orchestration, command parsing, config loading, output routing, report assembly, diagnostics metadata path assembly, and correction-ledger construction. This is acceptable for a small CLI, but two pieces are drifting beyond command responsibility:
- diagnostics artifact path metadata belongs closer to `internal/core/diagnostics`;
- report assembly and correction-ledger mapping belong closer to `internal/core/reporting` or a narrow reporting adapter package.
Recommended home:
- `internal/core/diagnostics`: artifact constants and diagnostics metadata path construction.
- `internal/core/reporting`: pure mapping from runner/config/diagnostics state into report payloads.
- `internal/cli`: command parsing, invocation wiring, exit codes, stdout/stderr behavior.
### Config validation lacks catalog ownership
`internal/core/config` currently validates only generic module list shape and hardcodes output schema keys. Because modules and output schemas are public contract values, config validation should use a catalog owned by the relevant domain.
Recommended home:
- output schema validation: `internal/core/outputschema`;
- module key validation: a small catalog package or lower-level constants package importable by config, module factory, validator chains, and threshold lookup.
### Runner owns adapter shims between contracts and validator framework
`internal/framework/runner` contains `validationLLMClientAdapter` and `llmDiagnosticsWriterAdapter`. This is not a serious problem today because runner wires proposal and validation workflows. If these adapters grow, move them to `internal/framework/validators` or a small integration package so runner remains focused on orchestration.
### LLM malformed-output policy is spread across callers
The LLM adapter emits the errors, while proposal generation and validators classify them by message text. The policy decision is caller-specific, but the classification should live with the LLM/framework error type.
## 6. Path, key, and naming construction review
Centralized enough:
- LLM diagnostics artifact suffixes and stage sanitization are centralized in `internal/framework/llm/diagnostics.go`.
- Output file writing is routed through `internal/core/io.WriteFile`.
- Run directories are created in `internal/core/diagnostics.NewRunDirectory`.
Needs cleanup:
- Core diagnostics artifact names are repeated between `RunDirectory` writer methods and `buildProcessReport`.
- `utilization-diagnostics.json` and `correction-ledger.json` are raw strings in both success and failure paths.
- Proposal and validator diagnostics subdirectory construction repeats `filepath.Join(diagnosticsDir, moduleInstance)`.
- Proposal and validator stage names are manually formatted in multiple packages.
- Module keys are repeated across config defaults, module factory, validator chains, confidence threshold lookup, and module implementations.
- Output schema names are repeated between config validation and `outputschema`.
Recommendation:
- Start with artifact constants and metadata helpers because that is the lowest-risk path/key cleanup.
- Then centralize stage-name helpers without changing current production naming.
- Defer any broader "path manager" abstraction.
## 7. Resolution and catalog review
Modules:
- Runtime module construction has a production registry in `internal/framework/modules`.
- Instance naming for repeated modules is centralized in `contracts.ResolveModuleRunSpecs`.
- Unknown module failure exists in the factory, but config validation does not catch unknown modules.
- Built-in validator chain resolution separately maps module key to validator keys.
Output schemas:
- Encoding is centralized in `internal/core/outputschema`.
- Validation is duplicated in config.
Prompts:
- Prompt asset lookup and metadata are centralized in `internal/prompts`.
- Prompt metadata map construction is repeated at call sites.
- Prompt source selection is intentionally built-in only and should remain that way for 1.0.
Validators:
- Validator construction is package-owned under `internal/validators`.
- Chains are centralized in `internal/validators/chains.go`.
- Execution class metadata exists, but reporting/correction-ledger classification does not fully use it.
Schemas:
- Transcript and glossary parsing/validation are centralized in `internal/core/schema`.
- Structured LLM response schemas are centralized in `internal/framework/responseschema`.
- Output schema registry and response schema registry are appropriately separate.
Recommendation:
- Introduce only small catalog helpers for module keys, output schema keys, prompt metadata maps, response schema metadata maps, and validator execution class.
- Avoid user-configurable modules, validators, prompts, or schemas before 1.0 unless already planned elsewhere.
## 8. Config and command-loading review
Consistent behavior:
- The documented precedence for `process` is implemented: defaults, file config, environment, CLI.
- `config print-effective` intentionally omits CLI process flags and uses defaults, file config, and environment.
- `config validate` intentionally requires `--config` and does not require transcript/glossary inputs.
- Missing explicit config paths are hard failures; missing default paths are non-fatal.
- Environment parsing and CLI parsing both preserve legacy total-concurrency alias behavior.
Likely accidental or high-risk differences:
- Unsupported module names pass `Config.Validate` and `audita config validate`.
- Output schema support is duplicated instead of delegated to the output schema registry.
- Config path resolution lives in CLI even though it is part of config behavior.
Intentional differences:
- File config resolves `api_key_env`; env and CLI set direct API key values.
- `OPENROUTER_API_KEY` is an environment fallback only for the primary LLM.
- `transcript-description` has CLI/config support but no `AUDITA_*` environment variable, matching documentation.
Recommendation:
- Build a shared effective config context helper and keep source-specific parsing semantics explicit.
- Tighten catalog validation before 1.0 if unknown modules are not meant to be accepted.
## 9. State, manifest, or progress handling review
Audita does not currently have a manifest/checkpoint/resume model. State is per-run diagnostics and report artifacts.
Consistent behavior:
- `process` creates one diagnostics run directory when diagnostics initialization succeeds.
- Failures after run-dir creation write `error.log`, best-effort report artifacts, and retain diagnostics.
- Success writes optional `--report-json`, run-dir `report.json`, utilization diagnostics, and correction ledger.
- Retention is centralized in `diagnostics.ShouldRetainRunDirectory`.
- There is no resume/retry/force behavior to preserve.
Drift risks:
- Success and failure paths both write utilization and correction-ledger artifacts with duplicated raw filenames.
- Report diagnostics metadata is assembled independently from the run-directory writer methods.
- Retention mode `never` currently still retains successful run directories in `ShouldRetainRunDirectory`, which may be intentional per tests or a naming/documentation mismatch. Do not change it in a dedup pass without first confirming semantics.
Recommendation:
- Centralize artifact names and report metadata path construction.
- Keep retention behavior unchanged unless a separate bug review confirms the intended meaning of `never`.
## 10. Refactors to avoid before 1.0
- Do not introduce a generic workflow engine. The current sequential runner is clear and explicit.
- Do not add a plugin architecture for modules, validators, prompts, or schemas before 1.0.
- Do not redesign the CLI or replace `flag` with a larger framework only for deduplication.
- Do not collapse all config source parsing into a reflection-based mapper; source semantics differ intentionally.
- Do not merge module packages into one generic module type. Keep domain-specific prompt assets, keys, validator chains, and replacement policies visible.
- Do not rewrite diagnostics or reporting schemas broadly. Centralize names and mapping helpers first.
- Do not change diagnostics stage names casually; they affect artifact filenames and debugging workflows.
- Do not consolidate deterministic and LLM validator behavior just because both return decisions. Their failure and batching semantics differ.
- Do not generalize transcript/glossary schema parsing into a broad schema framework.
- Do not reduce duplicated tests where the duplication protects distinct public command/module behavior.
## 11. Recommended implementation sequence
1. Centralize diagnostics artifact constants and diagnostics metadata path construction.
2. Centralize output schema validation through `internal/core/outputschema`.
3. Introduce a small module key catalog and use it in config validation, module factory, validator chains, and threshold lookup.
4. Add an effective config loading context helper for defaults + file + env, then update `process` and `config print-effective`.
5. Extract shared module proposal plumbing and prompt transcript-section payload construction.
6. Centralize prompt metadata and response schema metadata map construction.
7. Centralize validator execution-class lookup and update correction-ledger classification.
8. Centralize malformed structured-output classification through a typed/shared LLM error helper.
9. Add or consolidate focused test helpers for module LLM fakes, diagnostics assertions, and fixture paths.
10. Do a final dead-code and legacy sweep for redundant helper fields such as unused validator definition metadata.
Each item can be a separate commit with package-level tests and at least one CLI regression where public behavior is involved.
## 12. Test strategy
Tests to add before refactoring:
- `internal/core/config`: unknown module key fails validation, if unsupported modules are not intended to be accepted.
- `internal/core/config`: every output schema registry key validates through config.
- `internal/core/diagnostics`: report metadata paths match run-directory artifact names.
- `internal/validators`: validator class by key/instance is consistent for all registered validators.
- `internal/framework/llm`: shared malformed structured-output classifier covers all current adapter malformed errors.
Tests to add during refactoring:
- `internal/framework/promptcontext`: transcript section prompt payload preserves IDs, speaker, timestamps, text, and categories.
- `internal/framework/proposal_generation`: shared module proposal helper preserves current stage name, diagnostics dir, schema metadata, and malformed-output warning behavior.
- `internal/cli`: `process` and `config print-effective` share defaults+file+env behavior.
- `internal/cli`: `config validate` remains file-only and does not read env overrides.
- `internal/cli`: correction ledger classifies deterministic and LLM validator decisions through canonical metadata.
Existing tests to run after each cleanup:
- `go test ./internal/core/config ./internal/core/outputschema`
- `go test ./internal/core/diagnostics ./internal/core/reporting`
- `go test ./internal/framework/proposal_generation ./internal/framework/validators ./internal/framework/runner`
- `go test ./internal/validators/...`
- `go test ./internal/modules/...`
- `go test ./internal/cli ./cmd/audita`
- Run `go test ./...` before merging a multi-package cleanup.
Validation note:
- During this report-only pass, no full test suite was run. A lightweight `go list ./...` completed package listing but emitted a sandbox warning while trying to write the Go module stat cache outside the repository.
## 13. Appendix: findings not worth acting on
### Separate module packages
The four production module packages contain visible repetition, but keeping separate packages is useful. The module domains, prompt assets, validator chains, and tests are distinct enough that a single generic module package would hide important behavior.
Do not refactor now beyond shared proposal/prompt plumbing.
### Report type duplication between runner and reporting
`runner.ModuleResult` and `reporting.ModuleReport` look similar. Keeping separate runtime and public report shapes is reasonable because runner owns execution state and reporting owns serialized public schema.
Only centralize mapping helpers; do not merge the types.
### Transcript and glossary parsing stay separate
Transcript JSON and glossary YAML parsing have different formats, validation rules, and error messages. There is no useful shared parser abstraction to extract.
### Response schema registry and output schema registry stay separate
Structured LLM response schemas and transcript output schemas are both "schemas", but they serve different users and have different lifecycles. Do not combine their registries.
### `flag` package usage
The CLI command surface is small. Replacing `flag` with a larger CLI framework would not pay for itself before 1.0.
### Local test duplication that protects public behavior
Some test duplication in CLI, subprocess, parity, and release fixtures is intentional. These tests exercise different public surfaces and should remain explicit even if helpers are shared.
### Filesystem state as diagnostics state
Audita has no resume/checkpoint semantics. Treating diagnostics artifacts as filesystem outputs is currently acceptable. A manifest system would be speculative before there is a resume or audit workflow that needs it.

View File

@@ -0,0 +1,329 @@
# Pre-1.0 Deduplication Implementation Plan
This plan turns `docs/roadmap/audit.md` into staged, prompt-sized cleanup work for an LLM coding agent. Each stage should be implemented in order and kept small enough to review as an independent commit.
## Operating rules
- Read `docs/roadmap/audit.md` before starting any stage.
- Preserve public CLI, report, diagnostics, config precedence, prompt metadata, and output-schema behavior unless a stage explicitly calls out an intended behavior change.
- Keep the four production module packages separate: `glossary`, `homophones`, `spoken_word`, and `grammar`.
- Do not introduce plugin systems, generic workflow engines, broad CLI framework rewrites, reflection-heavy config mappers, or merged module packages.
- Prefer narrow helpers, catalogs, constants, and pure mapping functions over broad abstractions.
- Run the targeted tests listed in each stage before moving to the next stage.
- Run `go test ./...` before declaring the full sequence complete.
- Ignore unrelated worktree changes, including the existing deletion of `docs/roadmap/publish.md`, unless the user explicitly asks to handle them.
- Do not reduce parity, release-fixture, subprocess, or module-specific behavior coverage while consolidating helpers.
## Stages
### Stage 1: Diagnostics artifact constants and metadata paths
Goal:
- Centralize diagnostics artifact names and report diagnostics metadata path construction without changing any filenames or report fields.
Key edits:
- Define constants in `internal/core/diagnostics` for:
- `source-transcript.json`
- `source-transcript-parsed.json`
- `normalized-transcript.json`
- `normalization-summary.json`
- `chunking-summary.json`
- `utilization-diagnostics.json`
- `correction-ledger.json`
- `invocation.json`
- `effective-config.json`
- `report.json`
- `error.log`
- Add a diagnostics helper that builds `reporting.DiagnosticsMetadata` from a run directory path and failure/success status.
- Update `RunDirectory` methods to use the constants.
- Update CLI report assembly and utilization/correction-ledger writes to use the constants/helper instead of raw strings.
Behavior changes:
- None. All artifact names, report JSON keys, and path values must remain byte-for-byte compatible except for normal timestamp/order differences in existing outputs.
Tests:
- Add or update `internal/core/diagnostics` tests proving metadata helper paths match the artifact constants.
- Run `go test ./internal/core/diagnostics ./internal/core/reporting ./internal/cli`.
- Run any existing CLI report/diagnostics tests touched by this stage.
Acceptance criteria:
- No raw core diagnostics artifact filename strings remain in CLI report metadata assembly.
- Existing success and failure reports still point to files that are actually written.
- Retention behavior is unchanged.
### Stage 2: Output schema validation and module catalog
Goal:
- Move public key validation to small canonical catalogs so config validation, runtime resolution, and factory behavior cannot drift.
Key edits:
- Add `SupportedKeys`, `IsSupported`, or an equivalent validation helper to `internal/core/outputschema`.
- Update `config.Validate` to use `internal/core/outputschema` for output schema validation.
- Add a small canonical module key catalog that is importable by:
- `internal/core/config`
- `internal/framework/modules`
- `internal/validators`
- `internal/framework/validators`
- Use the module catalog for default module key constants, known-key checks, validator chain keys, and confidence-threshold lookup.
- Keep module construction in `internal/framework/modules`; the catalog must not construct modules.
Behavior changes:
- Intended behavior change: unsupported configured module keys should fail during config validation, including `audita config validate`.
- Repeated supported module keys remain valid.
- Output schema behavior remains unchanged for `bare-segments`, `audita-v1`, and unsupported names.
Tests:
- Add `internal/core/config` tests for unsupported module keys and repeated supported module keys.
- Add config validation tests that every supported output schema validates.
- Add or update output schema registry tests for supported and unsupported schemas.
- Update module registry and validator chain tests to use the shared catalog where appropriate.
- Run `go test ./internal/core/config ./internal/core/outputschema ./internal/framework/modules ./internal/framework/validators ./internal/validators/... ./internal/cli`.
Acceptance criteria:
- Unknown modules fail before runner setup in config validation paths.
- No duplicated hardcoded output schema support list remains in config validation.
- No import cycle is introduced.
### Stage 3: Effective config loading context
Goal:
- Centralize config path resolution and defaults+file+env loading while keeping command-specific CLI overrides explicit.
Key edits:
- Move config path resolution from `internal/cli` into `internal/core/config` or add an equivalent exported helper there.
- Add an effective config loader that returns:
- effective `config.Config`
- config path
- config source (`flag`, `env`, `default`, or empty)
- config version pointer when a file was loaded
- Use the shared loader in `audita process` before applying CLI overrides.
- Use the shared loader in `audita config print-effective`.
- Keep `audita config validate` as file-only: load file, apply to defaults, validate, and do not apply environment overrides.
Behavior changes:
- None. Preserve existing precedence:
- `process`: defaults, file config, environment, CLI flags
- `config print-effective`: defaults, file config, environment
- `config validate`: file config applied to defaults only
- Preserve explicit config path failure behavior and missing default path non-fatal behavior.
Tests:
- Add table-driven config loader tests for:
- explicit `--config`
- `AUDITA_CONFIG`
- default search paths
- missing explicit path
- missing env path
- missing default paths
- Add or update CLI tests proving `process` and `config print-effective` share file+env behavior.
- Add or update CLI tests proving `config validate` ignores environment overrides.
- Run `go test ./internal/core/config ./internal/cli ./cmd/audita`.
Acceptance criteria:
- Config precedence is unchanged.
- Config source/path/version metadata in invocation and reports is unchanged.
- Config command stdout/stderr and exit-code behavior is unchanged except for the intended unknown-module validation from Stage 2.
### Stage 4: Prompt/schema metadata and stage-name helpers
Goal:
- Centralize diagnostics-visible metadata and stage-name construction without changing production diagnostics names.
Key edits:
- Add a helper or method in `internal/prompts` that returns the stable prompt metadata diagnostics shape currently expanded by call sites.
- Add a helper or method in `internal/framework/responseschema` that returns the stable response schema metadata diagnostics shape currently expanded by call sites.
- Add shared proposal and validator stage-name helpers in the lowest package that avoids import cycles.
- Use the helpers in proposal generation, LLM validators, and production modules.
Behavior changes:
- None. Preserve current production stage names:
- module proposal stages keep their existing `proposal` naming form;
- validator batch stages keep their existing validator/batch naming form.
- Preserve all prompt metadata and response schema metadata field names and values.
Tests:
- Add prompt metadata helper tests covering every registered prompt.
- Add response schema metadata helper tests covering every registered response schema.
- Add stage-name helper tests for no-section, section, and validator batch cases.
- Run `go test ./internal/prompts ./internal/framework/responseschema ./internal/framework/proposal_generation ./internal/framework/validators ./internal/modules/...`.
Acceptance criteria:
- No manual prompt metadata map expansion remains in production module proposal plumbing.
- No duplicated response schema metadata map construction remains in proposal generation and LLM validators.
- Existing diagnostics fixture/path assertions still pass.
### Stage 5: Shared module proposal and prompt payload plumbing
Goal:
- Remove duplicated proposal execution and transcript-section prompt payload construction while preserving module-specific domain behavior.
Key edits:
- Add a narrow shared proposal execution helper, preferably in `internal/framework/proposal_generation` unless import cycles require a small module helper package.
- The helper should own:
- transcript description extraction from config;
- `GenerateCandidates` request construction;
- prompt metadata attachment;
- stage-name selection;
- conversion from generated corrections/warnings to `contracts.ProposalResult`.
- Add shared transcript-section prompt payload construction in `internal/framework/promptcontext`.
- Update each production module to provide only:
- module key;
- replacement policy;
- validator chain;
- prompt ID;
- domain-specific `BuildProposalMessages` call or message builder.
- Remove each module's redundant section transcript filtering if the runner already passes section-limited transcripts.
Behavior changes:
- None. Preserve module keys, replacement policies, validator chains, prompt IDs, diagnostics directories, proposal indexes, warning behavior, and correction mapping.
Tests:
- Add promptcontext tests for transcript section payload shape, empty transcript handling, section index, and category copying.
- Keep one module-specific prompt test per production module for domain wording and constraints.
- Add or update module proposal tests proving diagnostics are still written under the same module instance directory.
- Run `go test ./internal/framework/promptcontext ./internal/framework/proposal_generation ./internal/modules/... ./internal/cli`.
Acceptance criteria:
- Four production modules share proposal execution plumbing.
- Module packages remain separate and readable.
- CLI parity and release fixture behavior is unchanged.
### Stage 6: Validator classification and malformed LLM output policy
Goal:
- Use one source of truth for validator execution class and one shared classifier for malformed structured-output errors.
Key edits:
- Make validator execution class resolvable by stable validator key and by validator instance.
- Replace the correction-ledger hardcoded LLM-backed validator map with the canonical metadata source.
- Remove redundant validator metadata fields only after all call sites use the canonical source.
- Add a shared malformed structured-output classifier in `internal/framework/llm` or another low-level framework package.
- Update proposal generation and LLM validators to use the shared classifier while preserving their different handling outcomes.
Behavior changes:
- None. Proposal-generation malformed payloads still downgrade to warnings with zero proposals for affected sections.
- Validator malformed payloads still reject affected batches with warnings.
- Correction-ledger deterministic vs LLM validator sections should be unchanged for current validators.
Tests:
- Add validator metadata tests proving every registered validator has the expected execution class by key and instance.
- Add correction-ledger tests proving deterministic and LLM-backed decisions are classified through canonical metadata.
- Add shared malformed-output classifier tests covering current adapter malformed-output messages.
- Update proposal-generation and validator tests to assert representative malformed adapter errors are still downgraded.
- Run `go test ./internal/validators/... ./internal/framework/validators ./internal/framework/proposal_generation ./internal/framework/llm ./internal/cli`.
Acceptance criteria:
- No local hardcoded LLM-backed validator map remains in correction-ledger construction.
- Proposal-generation and validator malformed-output classifier lists cannot drift.
- Existing runner validator ordering is unchanged.
### Stage 7: Redaction and adapter workflow cleanup
Goal:
- Reduce duplicated secret extraction/redaction setup while preserving all no-secret-leak guarantees.
Key edits:
- Add a shared helper that extracts all configured LLM secret values from `config.Config`.
- Use the helper in proposal-generation diagnostics and validator diagnostics setup.
- Keep config structural redaction (`Config.Redacted`) separate from byte/string payload redaction.
- Keep adapter error redaction behavior compatible with current surfaced errors.
- Move runner adapter shims only if Stage 6 or this stage makes them materially larger; otherwise leave them in runner.
Behavior changes:
- None. Redaction token and no-secret-leak behavior remain unchanged.
Tests:
- Add or update tests proving proposal diagnostics, validator diagnostics, effective config artifacts, and surfaced adapter errors redact the same configured secrets.
- Keep existing subprocess no-secret-leak tests.
- Run `go test ./internal/core/config ./internal/framework/llm ./internal/framework/proposal_generation ./internal/framework/validators ./internal/cli ./cmd/audita`.
Acceptance criteria:
- Secret-list assembly is no longer duplicated between proposal and validator paths.
- No plaintext configured API key appears in diagnostics, reports, stdout, or stderr in existing redaction tests.
- No unrelated adapter behavior changes.
### Stage 8: Test helper cleanup and dead-code sweep
Goal:
- Consolidate test-only duplication and remove dead/redundant code left by prior stages.
Key edits:
- Consolidate package-local fake LLM clients, fixture readers, diagnostics glob helpers, and run-directory helpers where duplication is clear.
- Use cross-package test support only if it does not obscure test intent or introduce awkward imports.
- Remove redundant metadata fields, constants, or helper functions made obsolete by earlier stages.
- Keep module-specific prompt and behavior assertions local to each module package.
Behavior changes:
- None.
Tests:
- Run all package tests touched by helper cleanup.
- Run `go test ./internal/modules/... ./internal/framework/... ./internal/cli ./cmd/audita`.
- Run `go test ./...` before completing the full sequence.
Acceptance criteria:
- Test helpers are simpler without reducing coverage.
- No parity or release fixture assertions are removed unless replaced by equivalent or stronger assertions.
- No production behavior changes.
## Final verification
Before declaring the staged cleanup complete:
- Run:
- `go test ./internal/core/config ./internal/core/outputschema`
- `go test ./internal/core/diagnostics ./internal/core/reporting`
- `go test ./internal/framework/proposal_generation ./internal/framework/validators ./internal/framework/runner`
- `go test ./internal/validators/...`
- `go test ./internal/modules/...`
- `go test ./internal/cli ./cmd/audita`
- `go test ./...`
- Inspect `git diff` for accidental public CLI, config, report, diagnostics, prompt metadata, stage-name, or output-schema changes.
- Update docs only when behavior intentionally changes, especially the intended Stage 2 unknown-module validation change.
- Keep commits stage-sized and mention behavior-preservation tests in each commit message or PR description.
## Assumptions
- Unknown configured module keys should become config-validation failures before 1.0.
- Diagnostics filenames and stage names are public enough to preserve unless a stage explicitly says otherwise.
- Each stage should be implemented and reviewed separately.

View File

@@ -0,0 +1,121 @@
package cli
import (
"flag"
"gitea.maximumdirect.net/eric/audita/internal/core/config"
)
type processOverrideBinding func(*config.CLIOverrides, processFlags)
var processOverrideBindings = map[string]processOverrideBinding{
"modules": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ModulesCSV = flags.modules
},
"output-schema": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.OutputSchema = flags.outputSchema
},
"llm-api-key": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.PrimaryLLMAPIKey = flags.llmAPIKey
},
"validation-llm-api-key": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationLLMAPIKey = flags.validationLLMAPIKey
},
"model": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.PrimaryModel = flags.model
},
"validation-model": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationModel = flags.validationModel
},
"base-url": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.PrimaryBaseURL = flags.baseURL
},
"validation-base-url": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationBaseURL = flags.validationBaseURL
},
"llm-timeout-seconds": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.PrimaryLLMTimeoutSeconds = flags.llmTimeoutSeconds
},
"total-llm-concurrency": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.TotalLLMConcurrency = flags.totalLLMConcurrency
},
"proposal-llm-concurrency": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ProposalLLMConcurrency = flags.proposalLLMConcurrency
},
"llm-concurrency": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.PrimaryLLMConcurrency = flags.llmConcurrency
},
"validation-llm-timeout-seconds": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationLLMTimeoutSeconds = flags.validationLLMTimeoutSeconds
},
"max-retries": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.MaxRetries = flags.maxRetries
},
"validation-max-retries": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationMaxRetries = flags.validationMaxRetries
},
"validation-llm-concurrency": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationLLMConcurrency = flags.validationLLMConcurrency
},
"validation-max-prompt-tokens": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.ValidationMaxPromptTokens = flags.validationMaxPromptTokens
},
"max-section-tokens": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.MaxSectionTokens = flags.maxSectionTokens
},
"min-section-tokens": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.MinSectionTokens = flags.minSectionTokens
},
"target-sections": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.TargetSections = flags.targetSections
},
"glossary-confidence-threshold": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.GlossaryConfidenceThreshold = flags.glossaryConfidenceThreshold
},
"grammar-confidence-threshold": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.GrammarConfidenceThreshold = flags.grammarConfidenceThreshold
},
"homophones-confidence-threshold": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.HomophonesConfidenceThreshold = flags.homophonesConfidenceThreshold
},
"spoken-word-confidence-threshold": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.SpokenWordConfidenceThreshold = flags.spokenWordConfidenceThreshold
},
"normalize-max-segment-gap": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.NormalizeMaxSegmentGap = flags.normalizeMaxSegmentGap
},
"normalize-ellipsis-gap": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.NormalizeEllipsisGap = flags.normalizeEllipsisGap
},
"normalize-max-segment-duration": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.NormalizeMaxSegmentDuration = flags.normalizeMaxSegmentDuration
},
"normalize-max-segment-tokens": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.NormalizeMaxSegmentTokens = flags.normalizeMaxSegmentTokens
},
"transcript-description": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.TranscriptDescription = flags.transcriptDescription
},
"work-dir": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.WorkDir = flags.workDir
},
"work-dir-retention": func(overrides *config.CLIOverrides, flags processFlags) {
overrides.WorkDirRetention = flags.workDirRetention
},
}
func processCLIOverrides(fs *flag.FlagSet, flags processFlags) (config.CLIOverrides, bool) {
overrides := config.CLIOverrides{}
explicitModules := false
fs.Visit(func(f *flag.Flag) {
if f.Name == "modules" {
explicitModules = true
}
binding, ok := processOverrideBindings[f.Name]
if !ok {
return
}
binding(&overrides, flags)
})
return overrides, explicitModules
}

View File

@@ -0,0 +1,433 @@
package cli
import (
"io"
"reflect"
"strings"
"testing"
"gitea.maximumdirect.net/eric/audita/internal/core/config"
)
func TestProcessCLIOverridesMapsEveryConfigMutatingFlag(t *testing.T) {
tests := []struct {
name string
flagName string
value string
wantExplicitModules bool
assertOverrideFields func(t *testing.T, overrides config.CLIOverrides)
}{
{
name: "modules",
flagName: "modules",
value: "grammar,glossary",
wantExplicitModules: true,
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "ModulesCSV", overrides.ModulesCSV, "grammar,glossary")
},
},
{
name: "output schema",
flagName: "output-schema",
value: "audita-v1",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "OutputSchema", overrides.OutputSchema, "audita-v1")
},
},
{
name: "primary api key",
flagName: "llm-api-key",
value: "primary-key",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "PrimaryLLMAPIKey", overrides.PrimaryLLMAPIKey, "primary-key")
},
},
{
name: "validation api key",
flagName: "validation-llm-api-key",
value: "validation-key",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "ValidationLLMAPIKey", overrides.ValidationLLMAPIKey, "validation-key")
},
},
{
name: "primary model",
flagName: "model",
value: "primary-model",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "PrimaryModel", overrides.PrimaryModel, "primary-model")
},
},
{
name: "validation model",
flagName: "validation-model",
value: "validation-model",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "ValidationModel", overrides.ValidationModel, "validation-model")
},
},
{
name: "primary base url",
flagName: "base-url",
value: "https://primary.example.test",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "PrimaryBaseURL", overrides.PrimaryBaseURL, "https://primary.example.test")
},
},
{
name: "validation base url",
flagName: "validation-base-url",
value: "https://validation.example.test",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "ValidationBaseURL", overrides.ValidationBaseURL, "https://validation.example.test")
},
},
{
name: "primary timeout",
flagName: "llm-timeout-seconds",
value: "101",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "PrimaryLLMTimeoutSeconds", overrides.PrimaryLLMTimeoutSeconds, 101)
},
},
{
name: "total concurrency",
flagName: "total-llm-concurrency",
value: "5",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "TotalLLMConcurrency", overrides.TotalLLMConcurrency, 5)
},
},
{
name: "proposal concurrency",
flagName: "proposal-llm-concurrency",
value: "3",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "ProposalLLMConcurrency", overrides.ProposalLLMConcurrency, 3)
},
},
{
name: "legacy concurrency alias",
flagName: "llm-concurrency",
value: "4",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "PrimaryLLMConcurrency", overrides.PrimaryLLMConcurrency, 4)
},
},
{
name: "validation timeout",
flagName: "validation-llm-timeout-seconds",
value: "202",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "ValidationLLMTimeoutSeconds", overrides.ValidationLLMTimeoutSeconds, 202)
},
},
{
name: "max retries",
flagName: "max-retries",
value: "6",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "MaxRetries", overrides.MaxRetries, 6)
},
},
{
name: "validation max retries",
flagName: "validation-max-retries",
value: "7",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "ValidationMaxRetries", overrides.ValidationMaxRetries, 7)
},
},
{
name: "validation concurrency",
flagName: "validation-llm-concurrency",
value: "8",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "ValidationLLMConcurrency", overrides.ValidationLLMConcurrency, 8)
},
},
{
name: "validation max prompt tokens",
flagName: "validation-max-prompt-tokens",
value: "4096",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "ValidationMaxPromptTokens", overrides.ValidationMaxPromptTokens, 4096)
},
},
{
name: "max section tokens",
flagName: "max-section-tokens",
value: "9000",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "MaxSectionTokens", overrides.MaxSectionTokens, 9000)
},
},
{
name: "min section tokens",
flagName: "min-section-tokens",
value: "1000",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "MinSectionTokens", overrides.MinSectionTokens, 1000)
},
},
{
name: "target sections",
flagName: "target-sections",
value: "12",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "TargetSections", overrides.TargetSections, 12)
},
},
{
name: "glossary threshold",
flagName: "glossary-confidence-threshold",
value: "0.91",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "GlossaryConfidenceThreshold", overrides.GlossaryConfidenceThreshold, 0.91)
},
},
{
name: "grammar threshold",
flagName: "grammar-confidence-threshold",
value: "0.92",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "GrammarConfidenceThreshold", overrides.GrammarConfidenceThreshold, 0.92)
},
},
{
name: "homophones threshold",
flagName: "homophones-confidence-threshold",
value: "0.93",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "HomophonesConfidenceThreshold", overrides.HomophonesConfidenceThreshold, 0.93)
},
},
{
name: "spoken word threshold",
flagName: "spoken-word-confidence-threshold",
value: "0.94",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "SpokenWordConfidenceThreshold", overrides.SpokenWordConfidenceThreshold, 0.94)
},
},
{
name: "normalize max segment gap",
flagName: "normalize-max-segment-gap",
value: "1.2",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "NormalizeMaxSegmentGap", overrides.NormalizeMaxSegmentGap, 1.2)
},
},
{
name: "normalize ellipsis gap",
flagName: "normalize-ellipsis-gap",
value: "2.3",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "NormalizeEllipsisGap", overrides.NormalizeEllipsisGap, 2.3)
},
},
{
name: "normalize max segment duration",
flagName: "normalize-max-segment-duration",
value: "45.6",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertFloatOverride(t, "NormalizeMaxSegmentDuration", overrides.NormalizeMaxSegmentDuration, 45.6)
},
},
{
name: "normalize max segment tokens",
flagName: "normalize-max-segment-tokens",
value: "321",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertIntOverride(t, "NormalizeMaxSegmentTokens", overrides.NormalizeMaxSegmentTokens, 321)
},
},
{
name: "transcript description",
flagName: "transcript-description",
value: "podcast episode",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "TranscriptDescription", overrides.TranscriptDescription, "podcast episode")
},
},
{
name: "work dir",
flagName: "work-dir",
value: "/tmp/custom-audita",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "WorkDir", overrides.WorkDir, "/tmp/custom-audita")
},
},
{
name: "work dir retention",
flagName: "work-dir-retention",
value: "always",
assertOverrideFields: func(t *testing.T, overrides config.CLIOverrides) {
assertStringOverride(t, "WorkDirRetention", overrides.WorkDirRetention, "always")
},
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
fs, flags := newProcessFlagSet(config.Default(), io.Discard)
if err := fs.Parse([]string{"--" + tc.flagName, tc.value}); err != nil {
t.Fatalf("parse flag: %v", err)
}
overrides, explicitModules := processCLIOverrides(fs, flags)
if explicitModules != tc.wantExplicitModules {
t.Fatalf("explicitModules=%v, want %v", explicitModules, tc.wantExplicitModules)
}
tc.assertOverrideFields(t, overrides)
})
}
}
func TestProcessCLIOverridesIgnoresNonConfigFlags(t *testing.T) {
fs, flags := newProcessFlagSet(config.Default(), io.Discard)
if err := fs.Parse([]string{
"--config", "/tmp/config.yml",
"--glossary", "/tmp/glossary.yml",
"--output", "/tmp/output.json",
"--report-json", "/tmp/report.json",
}); err != nil {
t.Fatalf("parse flags: %v", err)
}
overrides, explicitModules := processCLIOverrides(fs, flags)
if explicitModules {
t.Fatal("non-config flags should not mark modules explicit")
}
assertNoCLIOverrides(t, overrides)
}
func TestNewProcessFlagSetDefaultsReflectEffectiveConfig(t *testing.T) {
cfg := config.Default()
cfg.Modules = []string{"grammar", "glossary"}
cfg.OutputSchema = "audita-v1"
cfg.PrimaryLLM.APIKey = "primary-key"
cfg.ValidationLLM.APIKey = "validation-key"
cfg.PrimaryLLM.Model = "primary-model"
cfg.ValidationLLM.Model = "validation-model"
cfg.PrimaryLLM.BaseURL = "https://primary.example.test"
cfg.ValidationLLM.BaseURL = "https://validation.example.test"
cfg.PrimaryLLM.TimeoutSeconds = 101
cfg.TotalLLMConcurrency = 5
cfg.ProposalLLMConcurrency = 3
cfg.PrimaryLLM.MaxRetries = 6
cfg.ValidationMaxPromptTokens = 4096
cfg.MaxSectionTokens = 9000
cfg.MinSectionTokens = 1000
cfg.Thresholds.Glossary = 0.91
cfg.Thresholds.Grammar = 0.92
cfg.Thresholds.Homophones = 0.93
cfg.Thresholds.SpokenWord = 0.94
cfg.Normalization.MaxSegmentGap = 1.2
cfg.Normalization.EllipsisGap = 2.3
cfg.Normalization.MaxSegmentDuration = 45.6
cfg.Normalization.MaxSegmentTokens = 321
cfg.TranscriptDescription = "podcast episode"
cfg.WorkDir = "/tmp/custom-audita"
cfg.WorkDirRetention = config.WorkDirRetentionAlways
validationTimeout := 202
validationRetries := 7
validationConcurrency := 8
targetSections := 12
cfg.ValidationLLM.TimeoutSeconds = &validationTimeout
cfg.ValidationLLM.MaxRetries = &validationRetries
cfg.ValidationLLMConcurrency = &validationConcurrency
cfg.TargetSections = &targetSections
_, flags := newProcessFlagSet(cfg, io.Discard)
assertStringOverride(t, "modules default", flags.modules, "grammar,glossary")
assertStringOverride(t, "output schema default", flags.outputSchema, "audita-v1")
assertStringOverride(t, "primary api key default", flags.llmAPIKey, "primary-key")
assertStringOverride(t, "validation api key default", flags.validationLLMAPIKey, "validation-key")
assertStringOverride(t, "primary model default", flags.model, "primary-model")
assertStringOverride(t, "validation model default", flags.validationModel, "validation-model")
assertStringOverride(t, "primary base url default", flags.baseURL, "https://primary.example.test")
assertStringOverride(t, "validation base url default", flags.validationBaseURL, "https://validation.example.test")
assertIntOverride(t, "primary timeout default", flags.llmTimeoutSeconds, 101)
assertIntOverride(t, "total concurrency default", flags.totalLLMConcurrency, 5)
assertIntOverride(t, "proposal concurrency default", flags.proposalLLMConcurrency, 3)
assertIntOverride(t, "legacy concurrency alias default", flags.llmConcurrency, 5)
assertIntOverride(t, "validation timeout default", flags.validationLLMTimeoutSeconds, validationTimeout)
assertIntOverride(t, "max retries default", flags.maxRetries, 6)
assertIntOverride(t, "validation max retries default", flags.validationMaxRetries, validationRetries)
assertIntOverride(t, "validation concurrency default", flags.validationLLMConcurrency, validationConcurrency)
assertIntOverride(t, "validation max prompt tokens default", flags.validationMaxPromptTokens, 4096)
assertIntOverride(t, "max section tokens default", flags.maxSectionTokens, 9000)
assertIntOverride(t, "min section tokens default", flags.minSectionTokens, 1000)
assertIntOverride(t, "target sections default", flags.targetSections, targetSections)
assertFloatOverride(t, "glossary threshold default", flags.glossaryConfidenceThreshold, 0.91)
assertFloatOverride(t, "grammar threshold default", flags.grammarConfidenceThreshold, 0.92)
assertFloatOverride(t, "homophones threshold default", flags.homophonesConfidenceThreshold, 0.93)
assertFloatOverride(t, "spoken word threshold default", flags.spokenWordConfidenceThreshold, 0.94)
assertFloatOverride(t, "normalize max segment gap default", flags.normalizeMaxSegmentGap, 1.2)
assertFloatOverride(t, "normalize ellipsis gap default", flags.normalizeEllipsisGap, 2.3)
assertFloatOverride(t, "normalize max segment duration default", flags.normalizeMaxSegmentDuration, 45.6)
assertIntOverride(t, "normalize max segment tokens default", flags.normalizeMaxSegmentTokens, 321)
assertStringOverride(t, "transcript description default", flags.transcriptDescription, "podcast episode")
assertStringOverride(t, "work dir default", flags.workDir, "/tmp/custom-audita")
assertStringOverride(t, "work dir retention default", flags.workDirRetention, "always")
}
func TestNewProcessFlagSetUsesFallbackDefaultsForUnsetOptionalConfig(t *testing.T) {
cfg := config.Default()
_, flags := newProcessFlagSet(cfg, io.Discard)
assertIntOverride(t, "validation timeout fallback", flags.validationLLMTimeoutSeconds, cfg.PrimaryLLM.TimeoutSeconds)
assertIntOverride(t, "validation retries fallback", flags.validationMaxRetries, cfg.PrimaryLLM.MaxRetries)
assertIntOverride(t, "validation concurrency fallback", flags.validationLLMConcurrency, cfg.TotalLLMConcurrency)
assertIntOverride(t, "target sections fallback", flags.targetSections, 0)
}
func assertStringOverride(t *testing.T, name string, got *string, want string) {
t.Helper()
if got == nil || *got != want {
t.Fatalf("%s=%v, want %q", name, pointerValue(got), want)
}
}
func assertIntOverride(t *testing.T, name string, got *int, want int) {
t.Helper()
if got == nil || *got != want {
t.Fatalf("%s=%v, want %d", name, pointerValue(got), want)
}
}
func assertFloatOverride(t *testing.T, name string, got *float64, want float64) {
t.Helper()
if got == nil || *got != want {
t.Fatalf("%s=%v, want %v", name, pointerValue(got), want)
}
}
func assertNoCLIOverrides(t *testing.T, overrides config.CLIOverrides) {
t.Helper()
value := reflect.ValueOf(overrides)
typ := value.Type()
for i := 0; i < value.NumField(); i++ {
field := value.Field(i)
if field.Kind() != reflect.Ptr {
t.Fatalf("unexpected non-pointer CLIOverrides field %s", typ.Field(i).Name)
}
if !field.IsNil() {
t.Fatalf("expected no CLI overrides, field %s was set", typ.Field(i).Name)
}
}
}
func pointerValue[T any](ptr *T) any {
if ptr == nil {
return "<nil>"
}
if stringer, ok := any(*ptr).(interface{ String() string }); ok {
return strings.TrimSpace(stringer.String())
}
return *ptr
}

View File

@@ -23,6 +23,7 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts" "gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/llm" "gitea.maximumdirect.net/eric/audita/internal/framework/llm"
"gitea.maximumdirect.net/eric/audita/internal/framework/modules" "gitea.maximumdirect.net/eric/audita/internal/framework/modules"
"gitea.maximumdirect.net/eric/audita/internal/framework/processreport"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation" "gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation"
"gitea.maximumdirect.net/eric/audita/internal/framework/runner" "gitea.maximumdirect.net/eric/audita/internal/framework/runner"
"gitea.maximumdirect.net/eric/audita/internal/framework/validators" "gitea.maximumdirect.net/eric/audita/internal/framework/validators"
@@ -374,30 +375,28 @@ func runProcess(args []string, stdout, stderr io.Writer) int {
fmt.Fprintf(stderr, "audita process: invalid CLI configuration: %v\n", err) fmt.Fprintf(stderr, "audita process: invalid CLI configuration: %v\n", err)
return 2 return 2
} }
configPath, configSource, err := resolveConfigPath(configPathOverride, configPathOverrideSet, os.LookupEnv) effectiveConfig, err := config.LoadEffectiveConfig(configPathOverride, configPathOverrideSet)
if err != nil { if err != nil {
fmt.Fprintf(stderr, "audita process: %v\n", err) var effectiveConfigErr *config.EffectiveConfigError
if errors.As(err, &effectiveConfigErr) {
switch effectiveConfigErr.Kind {
case config.EffectiveConfigErrorLoadFile, config.EffectiveConfigErrorApplyFile:
fmt.Fprintf(stderr, "audita process: invalid config file: %v\n", effectiveConfigErr)
case config.EffectiveConfigErrorApplyEnv:
fmt.Fprintf(stderr, "audita process: invalid environment configuration: %v\n", effectiveConfigErr)
default:
fmt.Fprintf(stderr, "audita process: %v\n", effectiveConfigErr)
}
} else {
fmt.Fprintf(stderr, "audita process: %v\n", err)
}
return 2 return 2
} }
cfg := config.Default() cfg := effectiveConfig.Config
var configVersion *int configPath := effectiveConfig.ConfigPath
if configPath != "" { configSource := effectiveConfig.ConfigSource
fileCfg, fileErr := config.LoadFileConfig(configPath) configVersion := effectiveConfig.ConfigVersion
if fileErr != nil {
fmt.Fprintf(stderr, "audita process: invalid config file: %v\n", fileErr)
return 2
}
if applyErr := cfg.ApplyFileConfig(fileCfg); applyErr != nil {
fmt.Fprintf(stderr, "audita process: invalid config file: %v\n", applyErr)
return 2
}
configVersion = &fileCfg.Version
}
if err := cfg.ApplyEnvOverrides(); err != nil {
fmt.Fprintf(stderr, "audita process: invalid environment configuration: %v\n", err)
return 2
}
fs, pFlags := newProcessFlagSet(cfg, stderr) fs, pFlags := newProcessFlagSet(cfg, stderr)
@@ -421,75 +420,7 @@ func runProcess(args []string, stdout, stderr io.Writer) int {
return 2 return 2
} }
overrides := config.CLIOverrides{} overrides, explicitModules := processCLIOverrides(fs, pFlags)
explicitModules := false
fs.Visit(func(f *flag.Flag) {
switch f.Name {
case "modules":
explicitModules = true
overrides.ModulesCSV = pFlags.modules
case "output-schema":
overrides.OutputSchema = pFlags.outputSchema
case "llm-api-key":
overrides.PrimaryLLMAPIKey = pFlags.llmAPIKey
case "validation-llm-api-key":
overrides.ValidationLLMAPIKey = pFlags.validationLLMAPIKey
case "model":
overrides.PrimaryModel = pFlags.model
case "validation-model":
overrides.ValidationModel = pFlags.validationModel
case "base-url":
overrides.PrimaryBaseURL = pFlags.baseURL
case "validation-base-url":
overrides.ValidationBaseURL = pFlags.validationBaseURL
case "llm-timeout-seconds":
overrides.PrimaryLLMTimeoutSeconds = pFlags.llmTimeoutSeconds
case "total-llm-concurrency":
overrides.TotalLLMConcurrency = pFlags.totalLLMConcurrency
case "proposal-llm-concurrency":
overrides.ProposalLLMConcurrency = pFlags.proposalLLMConcurrency
case "llm-concurrency":
overrides.PrimaryLLMConcurrency = pFlags.llmConcurrency
case "validation-llm-timeout-seconds":
overrides.ValidationLLMTimeoutSeconds = pFlags.validationLLMTimeoutSeconds
case "max-retries":
overrides.MaxRetries = pFlags.maxRetries
case "validation-max-retries":
overrides.ValidationMaxRetries = pFlags.validationMaxRetries
case "validation-llm-concurrency":
overrides.ValidationLLMConcurrency = pFlags.validationLLMConcurrency
case "validation-max-prompt-tokens":
overrides.ValidationMaxPromptTokens = pFlags.validationMaxPromptTokens
case "max-section-tokens":
overrides.MaxSectionTokens = pFlags.maxSectionTokens
case "min-section-tokens":
overrides.MinSectionTokens = pFlags.minSectionTokens
case "target-sections":
overrides.TargetSections = pFlags.targetSections
case "glossary-confidence-threshold":
overrides.GlossaryConfidenceThreshold = pFlags.glossaryConfidenceThreshold
case "grammar-confidence-threshold":
overrides.GrammarConfidenceThreshold = pFlags.grammarConfidenceThreshold
case "homophones-confidence-threshold":
overrides.HomophonesConfidenceThreshold = pFlags.homophonesConfidenceThreshold
case "spoken-word-confidence-threshold":
overrides.SpokenWordConfidenceThreshold = pFlags.spokenWordConfidenceThreshold
case "normalize-max-segment-gap":
overrides.NormalizeMaxSegmentGap = pFlags.normalizeMaxSegmentGap
case "normalize-ellipsis-gap":
overrides.NormalizeEllipsisGap = pFlags.normalizeEllipsisGap
case "normalize-max-segment-duration":
overrides.NormalizeMaxSegmentDuration = pFlags.normalizeMaxSegmentDuration
case "normalize-max-segment-tokens":
overrides.NormalizeMaxSegmentTokens = pFlags.normalizeMaxSegmentTokens
case "transcript-description":
overrides.TranscriptDescription = pFlags.transcriptDescription
case "work-dir":
overrides.WorkDir = pFlags.workDir
case "work-dir-retention":
overrides.WorkDirRetention = pFlags.workDirRetention
}
})
if err := cfg.ApplyCLIOverrides(overrides); err != nil { if err := cfg.ApplyCLIOverrides(overrides); err != nil {
fmt.Fprintf(stderr, "audita process: invalid CLI configuration: %v\n", err) fmt.Fprintf(stderr, "audita process: invalid CLI configuration: %v\n", err)
@@ -530,12 +461,15 @@ func runProcess(args []string, stdout, stderr io.Writer) int {
if runErr != nil { if runErr != nil {
if runDir != nil && runOutput != nil { if runDir != nil && runOutput != nil {
if runOutput.Utilization != nil { if runOutput.Utilization != nil {
_ = runDir.WriteJSONArtifact("utilization-diagnostics.json", runOutput.Utilization) _ = runDir.WriteJSONArtifact(diagnostics.ArtifactUtilizationSummary, runOutput.Utilization)
} }
_ = runDir.WriteJSONArtifact("correction-ledger.json", buildCorrectionLedger(runDir.Path(), runOutput)) _ = runDir.WriteJSONArtifact(diagnostics.ArtifactCorrectionLedger, processreport.BuildCorrectionLedger(processreport.CorrectionLedgerInput{
RunDirectoryPath: runDir.Path(),
RunOutput: runOutput,
}))
} }
errorPhase, errorMessage := extractErrorPhase(runErr) errorPhase, errorMessage := extractErrorPhase(runErr)
report := buildProcessReport("failed", inv, runDir, startedAt, completedAt, errorMessage, errorPhase, nil, nil, runOutput) report := processreport.Build(processReportInput("failed", inv, runDir, startedAt, completedAt, errorMessage, errorPhase, nil, nil, runOutput))
if strings.TrimSpace(inv.ReportJSONPath) != "" { if strings.TrimSpace(inv.ReportJSONPath) != "" {
if err := reporting.WriteProcessReport(inv.ReportJSONPath, report); err != nil { if err := reporting.WriteProcessReport(inv.ReportJSONPath, report); err != nil {
@@ -559,12 +493,15 @@ func runProcess(args []string, stdout, stderr io.Writer) int {
if runDir != nil && runOutput != nil { if runDir != nil && runOutput != nil {
if runOutput.Utilization != nil { if runOutput.Utilization != nil {
_ = runDir.WriteJSONArtifact("utilization-diagnostics.json", runOutput.Utilization) _ = runDir.WriteJSONArtifact(diagnostics.ArtifactUtilizationSummary, runOutput.Utilization)
} }
_ = runDir.WriteJSONArtifact("correction-ledger.json", buildCorrectionLedger(runDir.Path(), runOutput)) _ = runDir.WriteJSONArtifact(diagnostics.ArtifactCorrectionLedger, processreport.BuildCorrectionLedger(processreport.CorrectionLedgerInput{
RunDirectoryPath: runDir.Path(),
RunOutput: runOutput,
}))
} }
report := buildProcessReport("success", inv, runDir, startedAt, completedAt, "", "", normSummary, chunkSummary, runOutput) report := processreport.Build(processReportInput("success", inv, runDir, startedAt, completedAt, "", "", normSummary, chunkSummary, runOutput))
if strings.TrimSpace(inv.ReportJSONPath) != "" { if strings.TrimSpace(inv.ReportJSONPath) != "" {
if err := reporting.WriteProcessReport(inv.ReportJSONPath, report); err != nil { if err := reporting.WriteProcessReport(inv.ReportJSONPath, report); err != nil {
@@ -579,21 +516,11 @@ func runProcess(args []string, stdout, stderr io.Writer) int {
} }
} }
hasSkippedCorrections := false
if runOutput != nil {
for _, mr := range runOutput.ModuleResults {
if len(mr.SkippedChanges) > 0 || len(mr.ValidatorRejected) > 0 {
hasSkippedCorrections = true
break
}
}
}
if runDir != nil { if runDir != nil {
_ = runDir.WriteReport(report) _ = runDir.WriteReport(report)
if err := runDir.ApplyRetention(diagnostics.RetentionDecisionInput{ if err := runDir.ApplyRetention(diagnostics.RetentionDecisionInput{
RunSucceeded: true, RunSucceeded: true,
HasSkippedCorrections: hasSkippedCorrections, HasSkippedCorrections: processreport.HasSkippedCorrections(runOutput),
}); err != nil { }); err != nil {
fmt.Fprintf(stderr, "audita process: failed to apply work-dir retention: %v\n", err) fmt.Fprintf(stderr, "audita process: failed to apply work-dir retention: %v\n", err)
return 1 return 1
@@ -652,6 +579,10 @@ func runConfigValidate(args []string, stdout, stderr io.Writer) int {
fmt.Fprintf(stderr, "audita config validate: %v\n", err) fmt.Fprintf(stderr, "audita config validate: %v\n", err)
return 2 return 2
} }
if err := cfg.Validate(); err != nil {
fmt.Fprintf(stderr, "audita config validate: %v\n", err)
return 2
}
fmt.Fprintln(stdout, "config is valid") fmt.Fprintln(stdout, "config is valid")
return 0 return 0
} }
@@ -674,28 +605,13 @@ func runConfigPrintEffective(args []string, stdout, stderr io.Writer) int {
configPathValue := strings.TrimSpace(*configPath) configPathValue := strings.TrimSpace(*configPath)
configPathSet := configPathValue != "" configPathSet := configPathValue != ""
path, _, err := resolveConfigPath(configPathValue, configPathSet, os.LookupEnv) effectiveConfig, err := config.LoadEffectiveConfig(configPathValue, configPathSet)
if err != nil { if err != nil {
fmt.Fprintf(stderr, "audita config print-effective: %v\n", err) fmt.Fprintf(stderr, "audita config print-effective: %v\n", err)
return 2 return 2
} }
cfg := config.Default() cfg := effectiveConfig.Config
if path != "" {
fileCfg, fileErr := config.LoadFileConfig(path)
if fileErr != nil {
fmt.Fprintf(stderr, "audita config print-effective: %v\n", fileErr)
return 2
}
if applyErr := cfg.ApplyFileConfig(fileCfg); applyErr != nil {
fmt.Fprintf(stderr, "audita config print-effective: %v\n", applyErr)
return 2
}
}
if err := cfg.ApplyEnvOverrides(); err != nil {
fmt.Fprintf(stderr, "audita config print-effective: %v\n", err)
return 2
}
redacted := cfg.Redacted() redacted := cfg.Redacted()
out, err := json.MarshalIndent(redacted, "", " ") out, err := json.MarshalIndent(redacted, "", " ")
@@ -722,142 +638,28 @@ func extractErrorPhase(err error) (phase string, message string) {
return "", msg return "", msg
} }
func buildProcessReport(status string, inv processInvocation, runDir *diagnostics.RunDirectory, startedAt, completedAt time.Time, errorMessage string, errorPhase string, normalizationSummary *normalization.NormalizationSummary, chunkingSummary *chunking.Summary, runOutput *runner.RunOutput) reporting.ProcessReport { func processReportInput(status string, inv processInvocation, runDir *diagnostics.RunDirectory, startedAt, completedAt time.Time, errorMessage string, errorPhase string, normalizationSummary *normalization.NormalizationSummary, chunkingSummary *chunking.Summary, runOutput *runner.RunOutput) processreport.BuildInput {
report := reporting.ProcessReport{ runDirectoryPath := ""
ReportMetadata: reporting.ReportMetadata{
ReportSchemaName: reporting.DefaultProcessReportSchemaName,
ReportSchemaVersion: reporting.DefaultProcessReportSchemaVersion,
OutputSchema: inv.Config.OutputSchema,
ConfigVersion: inv.ConfigVersion,
},
Phase: "default_pipeline",
Status: status,
Operation: "process",
TranscriptPath: inv.TranscriptPath,
GlossaryPath: inv.GlossaryPath,
OutputPath: inv.OutputPath,
Modules: append([]string(nil), inv.Config.Modules...),
StartedAt: startedAt,
CompletedAt: &completedAt,
ErrorPhase: errorPhase,
}
if runDir != nil { if runDir != nil {
diagnosticsDir := runDir.Path() runDirectoryPath = runDir.Path()
report.Diagnostics = &reporting.DiagnosticsMetadata{
DirectoryPath: diagnosticsDir,
SourceTranscriptPath: filepath.Join(diagnosticsDir, "source-transcript.json"),
ParsedSourceTranscriptPath: filepath.Join(diagnosticsDir, "source-transcript-parsed.json"),
NormalizedTranscriptPath: filepath.Join(diagnosticsDir, "normalized-transcript.json"),
NormalizationSummaryPath: filepath.Join(diagnosticsDir, "normalization-summary.json"),
ChunkingSummaryPath: filepath.Join(diagnosticsDir, "chunking-summary.json"),
UtilizationSummaryPath: filepath.Join(diagnosticsDir, "utilization-diagnostics.json"),
CorrectionLedgerPath: filepath.Join(diagnosticsDir, "correction-ledger.json"),
InvocationMetadataPath: filepath.Join(diagnosticsDir, "invocation.json"),
RedactedEffectiveConfigPath: filepath.Join(diagnosticsDir, "effective-config.json"),
}
if status == "failed" {
report.Diagnostics.ErrorLogPath = filepath.Join(diagnosticsDir, "error.log")
}
} }
if errorMessage != "" { return processreport.BuildInput{
report.ErrorMessage = errorMessage Status: status,
TranscriptPath: inv.TranscriptPath,
GlossaryPath: inv.GlossaryPath,
OutputPath: inv.OutputPath,
Modules: inv.Config.Modules,
OutputSchema: inv.Config.OutputSchema,
ConfigVersion: inv.ConfigVersion,
StartedAt: startedAt,
CompletedAt: completedAt,
ErrorMessage: errorMessage,
ErrorPhase: errorPhase,
RunDirectoryPath: runDirectoryPath,
NormalizationSummary: normalizationSummary,
ChunkingSummary: chunkingSummary,
RunOutput: runOutput,
} }
if normalizationSummary != nil {
report.InputSegmentCount = &normalizationSummary.InputSegmentCount
report.NormalizedSegmentCount = &normalizationSummary.OutputSegmentCount
report.NormalizationMerges = &normalizationSummary.MergesPerformed
report.NormalizationIDReassignments = &normalizationSummary.IDsReassigned
report.NormalizationSkipped.DifferentSpeakers = &normalizationSummary.SkippedMerges.DifferentSpeakers
report.NormalizationSkipped.GapTooLarge = &normalizationSummary.SkippedMerges.GapTooLarge
report.NormalizationSkipped.DurationExceeded = &normalizationSummary.SkippedMerges.DurationExceeded
report.NormalizationSkipped.TokenLimitExceeded = &normalizationSummary.SkippedMerges.TokenLimitExceeded
}
if chunkingSummary != nil {
report.Chunking = &reporting.ChunkingSummary{
ChunkCount: chunkingSummary.ChunkCount,
MinEstimatedTokens: chunkingSummary.MinEstimatedTokens,
MaxEstimatedTokens: chunkingSummary.MaxEstimatedTokens,
TotalEstimatedTokens: chunkingSummary.TotalEstimatedTokens,
TargetSections: chunkingSummary.TargetSections,
MaxSectionTokens: chunkingSummary.MaxSectionTokens,
MinSectionTokens: chunkingSummary.MinSectionTokens,
}
}
report.ModulesSummary, report.ModuleResults = buildModuleReporting(runOutput)
return report
}
func buildModuleReporting(runOutput *runner.RunOutput) (*reporting.ModulesSummary, []reporting.ModuleReport) {
if runOutput == nil || len(runOutput.ModuleResults) == 0 {
return nil, nil
}
moduleReports := make([]reporting.ModuleReport, 0, len(runOutput.ModuleResults))
summary := &reporting.ModulesSummary{ModuleCount: len(runOutput.ModuleResults)}
for _, r := range runOutput.ModuleResults {
startedAt := r.StartedAt
completedAt := r.CompletedAt
moduleReports = append(moduleReports, reporting.ModuleReport{
ModuleKey: r.ModuleKey,
ModuleInstance: r.ModuleInstance,
ReplacementPolicy: string(r.ReplacementPolicy),
Status: r.Status,
ProposalCount: r.ProposalCount,
ValidatorDecisions: mapValidatorDecisions(r.ValidatorDecisions),
ValidatorRejected: mapValidatorRejected(r.ValidatorRejected),
AppliedChanges: r.AppliedChanges,
SkippedChanges: r.SkippedChanges,
ErrorMessage: r.ErrorMessage,
StartedAt: &startedAt,
CompletedAt: &completedAt,
})
summary.TotalAppliedChanges += len(r.AppliedChanges)
summary.TotalSkippedChanges += len(r.SkippedChanges) + len(r.ValidatorRejected)
if r.Status == runner.ModuleStatusFailed && summary.FailedModuleInstance == "" {
summary.FailedModuleInstance = r.ModuleInstance
}
}
return summary, moduleReports
}
func mapValidatorDecisions(in []runner.ValidatorDecisionRecord) []reporting.ValidatorDecisionReport {
if len(in) == 0 {
return nil
}
out := make([]reporting.ValidatorDecisionReport, len(in))
for i, d := range in {
out[i] = reporting.ValidatorDecisionReport{
ValidatorName: d.ValidatorName,
ProposalIndex: d.ProposalIndex,
Approved: d.Approved,
ReasonCode: d.ReasonCode,
Message: d.Message,
DiagnosticArtifactPath: d.DiagnosticArtifactPath,
}
}
return out
}
func mapValidatorRejected(in []runner.ValidatorRejectedChange) []reporting.ValidatorRejectedReport {
if len(in) == 0 {
return nil
}
out := make([]reporting.ValidatorRejectedReport, len(in))
for i, d := range in {
out[i] = reporting.ValidatorRejectedReport{
ValidatorName: d.ValidatorName,
ProposalIndex: d.ProposalIndex,
ModuleKey: d.ModuleKey,
ModuleInstance: d.ModuleInstance,
TargetSegmentID: d.TargetSegmentID,
OriginalText: d.OriginalText,
CorrectedText: d.CorrectedText,
ReasonCode: d.ReasonCode,
Message: d.Message,
}
}
return out
} }
type processFlags struct { type processFlags struct {
@@ -982,47 +784,6 @@ func findConfigPathOverride(args []string) (path string, set bool, err error) {
return "", false, nil return "", false, nil
} }
var statConfigPath = os.Stat
func resolveConfigPath(cliConfigPath string, cliConfigPathSet bool, lookup func(string) (string, bool)) (path string, source string, err error) {
if cliConfigPathSet {
path = strings.TrimSpace(cliConfigPath)
if path == "" {
return "", "", fmt.Errorf("--config requires a non-empty path")
}
if _, statErr := statConfigPath(path); statErr != nil {
if os.IsNotExist(statErr) {
return "", "", fmt.Errorf("config file not found: %s", path)
}
return "", "", fmt.Errorf("cannot access config file %s: %w", path, statErr)
}
return path, "flag", nil
}
if raw, ok := lookup("AUDITA_CONFIG"); ok {
path = strings.TrimSpace(raw)
if path == "" {
return "", "", fmt.Errorf("AUDITA_CONFIG must not be empty")
}
if _, statErr := statConfigPath(path); statErr != nil {
if os.IsNotExist(statErr) {
return "", "", fmt.Errorf("config file not found: %s", path)
}
return "", "", fmt.Errorf("cannot access config file %s: %w", path, statErr)
}
return path, "env", nil
}
for _, defaultPath := range config.DefaultConfigSearchPaths {
if _, statErr := statConfigPath(defaultPath); statErr == nil {
return defaultPath, "default", nil
} else if !os.IsNotExist(statErr) {
return "", "", fmt.Errorf("cannot access config file %s: %w", defaultPath, statErr)
}
}
return "", "", nil
}
func isHelpCommand(args []string) bool { func isHelpCommand(args []string) bool {
if len(args) == 0 { if len(args) == 0 {
return false return false

View File

@@ -21,11 +21,11 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/core/schema" "gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts" "gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/llm" "gitea.maximumdirect.net/eric/audita/internal/framework/llm"
"gitea.maximumdirect.net/eric/audita/internal/framework/modules"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation" "gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals" "gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/runner" "gitea.maximumdirect.net/eric/audita/internal/framework/runner"
"gitea.maximumdirect.net/eric/audita/internal/framework/validators" "gitea.maximumdirect.net/eric/audita/internal/framework/validators"
"gitea.maximumdirect.net/eric/audita/internal/testsupport"
) )
func TestRunRootHelp(t *testing.T) { func TestRunRootHelp(t *testing.T) {
@@ -99,66 +99,6 @@ func TestRunProcessHelpListsExpectedFlags(t *testing.T) {
} }
} }
func TestResolveConfigPathDefaultIgnoredWhenMissing(t *testing.T) {
lookup := func(string) (string, bool) { return "", false }
path, source, err := resolveConfigPath("", false, lookup)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if path != "" || source != "" {
t.Fatalf("expected no config path/source, got path=%q source=%q", path, source)
}
}
func TestResolveConfigPathDefaultPrefersUsrLocalOverEtc(t *testing.T) {
oldStat := statConfigPath
statConfigPath = func(path string) (os.FileInfo, error) {
if path == config.DefaultConfigPathUsrLocal || path == config.DefaultConfigPath {
return nil, nil
}
return nil, os.ErrNotExist
}
t.Cleanup(func() { statConfigPath = oldStat })
lookup := func(string) (string, bool) { return "", false }
path, source, err := resolveConfigPath("", false, lookup)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if source != "default" {
t.Fatalf("expected default source, got %q", source)
}
if path != config.DefaultConfigPathUsrLocal {
t.Fatalf("expected %q, got %q", config.DefaultConfigPathUsrLocal, path)
}
}
func TestResolveConfigPathDefaultFallsBackToEtc(t *testing.T) {
oldStat := statConfigPath
statConfigPath = func(path string) (os.FileInfo, error) {
if path == config.DefaultConfigPathUsrLocal {
return nil, os.ErrNotExist
}
if path == config.DefaultConfigPath {
return nil, nil
}
return nil, os.ErrNotExist
}
t.Cleanup(func() { statConfigPath = oldStat })
lookup := func(string) (string, bool) { return "", false }
path, source, err := resolveConfigPath("", false, lookup)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if source != "default" {
t.Fatalf("expected default source, got %q", source)
}
if path != config.DefaultConfigPath {
t.Fatalf("expected %q, got %q", config.DefaultConfigPath, path)
}
}
func TestRunConfigValidateSuccess(t *testing.T) { func TestRunConfigValidateSuccess(t *testing.T) {
var stdout bytes.Buffer var stdout bytes.Buffer
var stderr bytes.Buffer var stderr bytes.Buffer
@@ -210,6 +150,23 @@ func TestRunConfigValidateUnknownField(t *testing.T) {
} }
} }
func TestRunConfigValidateUnsupportedModuleKey(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
cfgPath := writeFile(t, "config.yml", "version: 1\npipeline:\n modules: [made_up]\n")
exitCode := Run([]string{"config", "validate", "--config", cfgPath}, &stdout, &stderr)
if exitCode == 0 {
t.Fatalf("expected failure for unsupported module key")
}
if stdout.Len() != 0 {
t.Fatalf("expected empty stdout on failure, got %q", stdout.String())
}
if !strings.Contains(stderr.String(), "unsupported module key") {
t.Fatalf("expected unsupported module key error, got %q", stderr.String())
}
}
func TestRunConfigPrintEffectiveOutputsRedactedJSON(t *testing.T) { func TestRunConfigPrintEffectiveOutputsRedactedJSON(t *testing.T) {
var stdout bytes.Buffer var stdout bytes.Buffer
var stderr bytes.Buffer var stderr bytes.Buffer
@@ -242,6 +199,45 @@ func TestRunConfigPrintEffectiveOutputsRedactedJSON(t *testing.T) {
} }
} }
func TestRunConfigPrintEffectiveAppliesFileThenEnvironment(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
cfgPath := writeFile(t, "config.yml", "version: 1\nllm:\n proposal:\n model: file-model\n")
t.Setenv("AUDITA_MODEL", "env-model")
exitCode := Run([]string{"config", "print-effective", "--config", cfgPath}, &stdout, &stderr)
if exitCode != 0 {
t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
}
var out struct {
PrimaryLLM struct {
Model string `json:"Model"`
} `json:"PrimaryLLM"`
}
if err := json.Unmarshal(stdout.Bytes(), &out); err != nil {
t.Fatalf("expected valid JSON output, got error: %v output=%q", err, stdout.String())
}
if out.PrimaryLLM.Model != "env-model" {
t.Fatalf("expected env model override in print-effective output, got %q", out.PrimaryLLM.Model)
}
}
func TestRunConfigValidateIgnoresEnvironmentOverrides(t *testing.T) {
var stdout bytes.Buffer
var stderr bytes.Buffer
cfgPath := writeFile(t, "config.yml", "version: 1\n")
t.Setenv("AUDITA_MODULES", "made_up")
exitCode := Run([]string{"config", "validate", "--config", cfgPath}, &stdout, &stderr)
if exitCode != 0 {
t.Fatalf("expected success because config validate is file-only, got %d stderr=%q", exitCode, stderr.String())
}
if !strings.Contains(stdout.String(), "config is valid") {
t.Fatalf("expected success message, got %q", stdout.String())
}
}
func TestRunConfigCommandDoesNotRequireTranscriptOrGlossary(t *testing.T) { func TestRunConfigCommandDoesNotRequireTranscriptOrGlossary(t *testing.T) {
var stdout bytes.Buffer var stdout bytes.Buffer
var stderr bytes.Buffer var stderr bytes.Buffer
@@ -369,8 +365,8 @@ diagnostics:
func TestRunProcessEnvOverridesConfigFile(t *testing.T) { func TestRunProcessEnvOverridesConfigFile(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{ processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{ "grammar": fakeModule{
key: "m", key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique, policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{ validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) { fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -392,7 +388,7 @@ func TestRunProcessEnvOverridesConfigFile(t *testing.T) {
cfgPath := writeFile(t, "config.yml", ` cfgPath := writeFile(t, "config.yml", `
version: 1 version: 1
pipeline: pipeline:
modules: [m] modules: [grammar]
llm: llm:
proposal: proposal:
model: file-model model: file-model
@@ -404,7 +400,7 @@ llm:
fixturePath("tiny_transcript.json"), fixturePath("tiny_transcript.json"),
"--glossary", fixturePath("tiny_glossary.yaml"), "--glossary", fixturePath("tiny_glossary.yaml"),
"--config", cfgPath, "--config", cfgPath,
"--modules", "m", "--modules", "grammar",
}, &stdout, &stderr) }, &stdout, &stderr)
if exitCode != 0 { if exitCode != 0 {
t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String()) t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
@@ -413,8 +409,8 @@ llm:
func TestRunProcessCLIOverridesEnvAndConfigFile(t *testing.T) { func TestRunProcessCLIOverridesEnvAndConfigFile(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{ processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{ "grammar": fakeModule{
key: "m", key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique, policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{ validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) { fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -436,7 +432,7 @@ func TestRunProcessCLIOverridesEnvAndConfigFile(t *testing.T) {
cfgPath := writeFile(t, "config.yml", ` cfgPath := writeFile(t, "config.yml", `
version: 1 version: 1
pipeline: pipeline:
modules: [m] modules: [grammar]
llm: llm:
proposal: proposal:
model: file-model model: file-model
@@ -448,7 +444,7 @@ llm:
fixturePath("tiny_transcript.json"), fixturePath("tiny_transcript.json"),
"--glossary", fixturePath("tiny_glossary.yaml"), "--glossary", fixturePath("tiny_glossary.yaml"),
"--config", cfgPath, "--config", cfgPath,
"--modules", "m", "--modules", "grammar",
"--model", "cli-model", "--model", "cli-model",
}, &stdout, &stderr) }, &stdout, &stderr)
if exitCode != 0 { if exitCode != 0 {
@@ -495,8 +491,8 @@ diagnostics:
func TestRunProcessTranscriptDescriptionCLIOverridesConfigFileContextDescription(t *testing.T) { func TestRunProcessTranscriptDescriptionCLIOverridesConfigFileContextDescription(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{ processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{ "grammar": fakeModule{
key: "m", key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique, policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{ validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) { fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -517,7 +513,7 @@ func TestRunProcessTranscriptDescriptionCLIOverridesConfigFileContextDescription
cfgPath := writeFile(t, "config.yml", ` cfgPath := writeFile(t, "config.yml", `
version: 1 version: 1
pipeline: pipeline:
modules: [m] modules: [grammar]
context: context:
description: "file transcript description" description: "file transcript description"
`) `)
@@ -528,7 +524,7 @@ context:
fixturePath("tiny_transcript.json"), fixturePath("tiny_transcript.json"),
"--glossary", fixturePath("tiny_glossary.yaml"), "--glossary", fixturePath("tiny_glossary.yaml"),
"--config", cfgPath, "--config", cfgPath,
"--modules", "m", "--modules", "grammar",
"--transcript-description", "cli transcript description", "--transcript-description", "cli transcript description",
}, &stdout, &stderr) }, &stdout, &stderr)
if exitCode != 0 { if exitCode != 0 {
@@ -692,8 +688,8 @@ func TestRunProcessCLIOverridesEnvironment(t *testing.T) {
func TestRunProcessTranscriptDescriptionDefaultEmpty(t *testing.T) { func TestRunProcessTranscriptDescriptionDefaultEmpty(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{ processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{ "grammar": fakeModule{
key: "m", key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique, policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{ validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) { fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -720,7 +716,7 @@ func TestRunProcessTranscriptDescriptionDefaultEmpty(t *testing.T) {
exitCode := Run([]string{ exitCode := Run([]string{
"process", transcriptPath, "process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"), "--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "m", "--modules", "grammar",
}, &stdout, &stderr) }, &stdout, &stderr)
if exitCode != 0 { if exitCode != 0 {
t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String()) t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
@@ -729,8 +725,8 @@ func TestRunProcessTranscriptDescriptionDefaultEmpty(t *testing.T) {
func TestRunProcessTranscriptDescriptionCLIOverrideAndTrim(t *testing.T) { func TestRunProcessTranscriptDescriptionCLIOverrideAndTrim(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{ processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{ "grammar": fakeModule{
key: "m", key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique, policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{ validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) { fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -757,7 +753,7 @@ func TestRunProcessTranscriptDescriptionCLIOverrideAndTrim(t *testing.T) {
exitCode := Run([]string{ exitCode := Run([]string{
"process", transcriptPath, "process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"), "--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "m", "--modules", "grammar",
"--transcript-description", " speaker background context ", "--transcript-description", " speaker background context ",
}, &stdout, &stderr) }, &stdout, &stderr)
if exitCode != 0 { if exitCode != 0 {
@@ -823,8 +819,8 @@ func TestRunProcessRejectsValidationConcurrencyAboveTotalConcurrency(t *testing.
func TestRunProcessTotalLLMConcurrencyDrivesEffectiveValidationConcurrencyWhenUnset(t *testing.T) { func TestRunProcessTotalLLMConcurrencyDrivesEffectiveValidationConcurrencyWhenUnset(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{ processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{ "grammar": fakeModule{
key: "m", key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique, policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{ validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) { fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -865,7 +861,7 @@ func TestRunProcessTotalLLMConcurrencyDrivesEffectiveValidationConcurrencyWhenUn
"--glossary", "--glossary",
fixturePath("tiny_glossary.yaml"), fixturePath("tiny_glossary.yaml"),
"--modules", "--modules",
"m", "grammar",
"--total-llm-concurrency", "--total-llm-concurrency",
"4", "4",
}, &stdout, &stderr) }, &stdout, &stderr)
@@ -908,8 +904,8 @@ func TestRunProcessRejectsProposalConcurrencyAboveTotalConcurrency(t *testing.T)
func TestRunProcessLegacyLLMConcurrencyAliasSetsTotalAndProposal(t *testing.T) { func TestRunProcessLegacyLLMConcurrencyAliasSetsTotalAndProposal(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{ processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{ "grammar": fakeModule{
key: "m", key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique, policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{ validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) { fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -944,7 +940,7 @@ func TestRunProcessLegacyLLMConcurrencyAliasSetsTotalAndProposal(t *testing.T) {
"--glossary", "--glossary",
fixturePath("tiny_glossary.yaml"), fixturePath("tiny_glossary.yaml"),
"--modules", "--modules",
"m", "grammar",
"--llm-concurrency", "--llm-concurrency",
"3", "3",
}, &stdout, &stderr) }, &stdout, &stderr)
@@ -955,8 +951,8 @@ func TestRunProcessLegacyLLMConcurrencyAliasSetsTotalAndProposal(t *testing.T) {
func TestRunProcessLLMConcurrencyFlagsOverrideEnvironment(t *testing.T) { func TestRunProcessLLMConcurrencyFlagsOverrideEnvironment(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{ processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{ "grammar": fakeModule{
key: "m", key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique, policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{ validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) { fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -997,7 +993,7 @@ func TestRunProcessLLMConcurrencyFlagsOverrideEnvironment(t *testing.T) {
"--glossary", "--glossary",
fixturePath("tiny_glossary.yaml"), fixturePath("tiny_glossary.yaml"),
"--modules", "--modules",
"m", "grammar",
"--total-llm-concurrency", "--total-llm-concurrency",
"4", "4",
"--proposal-llm-concurrency", "--proposal-llm-concurrency",
@@ -1010,8 +1006,8 @@ func TestRunProcessLLMConcurrencyFlagsOverrideEnvironment(t *testing.T) {
func TestRunProcessAcceptsLLMConcurrencyEnvironmentVariables(t *testing.T) { func TestRunProcessAcceptsLLMConcurrencyEnvironmentVariables(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{ processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{ "grammar": fakeModule{
key: "m", key: "grammar",
policy: proposals.ReplacementPolicyRequireUnique, policy: proposals.ReplacementPolicyRequireUnique,
validators: []contracts.Validator{ validators: []contracts.Validator{
fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) { fakeValidator{name: "capture-config", validateF: func(req contracts.ValidationRequest) (validators.Result, error) {
@@ -1052,7 +1048,7 @@ func TestRunProcessAcceptsLLMConcurrencyEnvironmentVariables(t *testing.T) {
"--glossary", "--glossary",
fixturePath("tiny_glossary.yaml"), fixturePath("tiny_glossary.yaml"),
"--modules", "--modules",
"m", "grammar",
}, &stdout, &stderr) }, &stdout, &stderr)
if exitCode != 0 { if exitCode != 0 {
t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String()) t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
@@ -1793,11 +1789,12 @@ type fakeModule struct {
func (m fakeModule) Key() string { return m.key } func (m fakeModule) Key() string { return m.key }
func (m fakeModule) ReplacementPolicy() proposals.ReplacementPolicy { return m.policy } func (m fakeModule) ReplacementPolicy() proposals.ReplacementPolicy { return m.policy }
func (m fakeModule) Validators() []contracts.Validator { return m.validators } func (m fakeModule) Validators() []contracts.Validator { return m.validators }
func (m fakeModule) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { func (m fakeModule) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
if m.proposeF == nil { if m.proposeF == nil {
return nil, nil return contracts.ProposalResult{}, nil
} }
return m.proposeF(req) proposalsOut, err := m.proposeF(req)
return contracts.ProposalResult{Proposals: proposalsOut}, err
} }
type fakeValidator struct { type fakeValidator struct {
@@ -1882,12 +1879,12 @@ func TestRunProcessInjectedFactoryExecutesRunnerAndReportsModules(t *testing.T)
return validators.Result{ValidatorName: "allow", Decisions: decisions}, nil return validators.Result{ValidatorName: "allow", Decisions: decisions}, nil
}} }}
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{ processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m1": fakeModule{key: "m1", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{allow}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { "glossary": fakeModule{key: "glossary", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{allow}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
return []proposals.CorrectionProposal{ return []proposals.CorrectionProposal{
{TargetSegmentID: 1, OriginalText: "Hello", CorrectedText: "Hi", Confidence: 1}, {TargetSegmentID: 1, OriginalText: "Hello", CorrectedText: "Hi", Confidence: 1},
}, nil }, nil
}}, }},
"m2": fakeModule{key: "m2", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{allow}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { "homophones": fakeModule{key: "homophones", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{allow}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
if req.WorkingTranscript.Segments[0].Text != "Hi world" { if req.WorkingTranscript.Segments[0].Text != "Hi world" {
t.Fatalf("expected module 2 to see module 1 changes, got %q", req.WorkingTranscript.Segments[0].Text) t.Fatalf("expected module 2 to see module 1 changes, got %q", req.WorkingTranscript.Segments[0].Text)
} }
@@ -1910,7 +1907,7 @@ func TestRunProcessInjectedFactoryExecutesRunnerAndReportsModules(t *testing.T)
exitCode := Run([]string{ exitCode := Run([]string{
"process", transcriptPath, "process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"), "--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "m1,m2", "--modules", "glossary,homophones",
"--output", outputPath, "--output", outputPath,
"--report-json", reportPath, "--report-json", reportPath,
"--work-dir", workDir, "--work-dir", workDir,
@@ -1969,7 +1966,7 @@ func TestRunProcessInjectedFactoryLLMValidatorResultsInReports(t *testing.T) {
}}, }},
} }
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{ processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{key: "m", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{llmValidator}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { "grammar": fakeModule{key: "grammar", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{llmValidator}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
return []proposals.CorrectionProposal{{TargetSegmentID: 1, OriginalText: "Hello", CorrectedText: "Hi", Confidence: 1}}, nil return []proposals.CorrectionProposal{{TargetSegmentID: 1, OriginalText: "Hello", CorrectedText: "Hi", Confidence: 1}}, nil
}}, }},
}} }}
@@ -1989,7 +1986,7 @@ func TestRunProcessInjectedFactoryLLMValidatorResultsInReports(t *testing.T) {
exitCode := Run([]string{ exitCode := Run([]string{
"process", transcriptPath, "process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"), "--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "m", "--modules", "grammar",
"--output", outputPath, "--output", outputPath,
"--report-json", reportPath, "--report-json", reportPath,
"--work-dir", workDir, "--work-dir", workDir,
@@ -2010,7 +2007,7 @@ func TestRunProcessInjectedFactoryLLMValidatorResultsInReports(t *testing.T) {
func TestRunProcessInjectedFactorySkippedKeepsAutoRetention(t *testing.T) { func TestRunProcessInjectedFactorySkippedKeepsAutoRetention(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{ processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m1": fakeModule{key: "m1", policy: proposals.ReplacementPolicyRequireUnique, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { "glossary": fakeModule{key: "glossary", policy: proposals.ReplacementPolicyRequireUnique, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
return []proposals.CorrectionProposal{ return []proposals.CorrectionProposal{
{TargetSegmentID: 1, OriginalText: "word", CorrectedText: "term", Confidence: 1}, {TargetSegmentID: 1, OriginalText: "word", CorrectedText: "term", Confidence: 1},
}, nil }, nil
@@ -2026,7 +2023,7 @@ func TestRunProcessInjectedFactorySkippedKeepsAutoRetention(t *testing.T) {
exitCode := Run([]string{ exitCode := Run([]string{
"process", transcriptPath, "process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"), "--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "m1", "--modules", "glossary",
"--work-dir", workDir, "--work-dir", workDir,
"--work-dir-retention", "auto", "--work-dir-retention", "auto",
}, &stdout, &stderr) }, &stdout, &stderr)
@@ -2040,7 +2037,7 @@ func TestRunProcessInjectedFactorySkippedKeepsAutoRetention(t *testing.T) {
func TestRunProcessInjectedFactoryFailureWritesFailedReport(t *testing.T) { func TestRunProcessInjectedFactoryFailureWritesFailedReport(t *testing.T) {
processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{ processModuleFactory = fakeModuleFactory{modules: map[string]contracts.TranscriptModule{
"m1": fakeModule{key: "m1", policy: proposals.ReplacementPolicyRequireUnique, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { "glossary": fakeModule{key: "glossary", policy: proposals.ReplacementPolicyRequireUnique, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
return nil, errors.New("test failure") return nil, errors.New("test failure")
}}, }},
}} }}
@@ -2056,7 +2053,7 @@ func TestRunProcessInjectedFactoryFailureWritesFailedReport(t *testing.T) {
exitCode := Run([]string{ exitCode := Run([]string{
"process", transcriptPath, "process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"), "--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "m1", "--modules", "glossary",
"--work-dir", workDir, "--work-dir", workDir,
"--work-dir-retention", "always", "--work-dir-retention", "always",
"--report-json", reportPath, "--report-json", reportPath,
@@ -2088,14 +2085,8 @@ func TestRunProcessInjectedFactoryFailureWritesFailedReport(t *testing.T) {
} }
} }
func TestRunProcessProductionRegistryUnsupportedModuleFailsCleanly(t *testing.T) { func TestRunProcessUnsupportedModuleFailsDuringConfigValidation(t *testing.T) {
cfg := modules.Dependencies{}
processModuleFactory = modules.NewFactory(cfg)
t.Cleanup(func() { processModuleFactory = nil })
var stdout, stderr bytes.Buffer var stdout, stderr bytes.Buffer
workDir := t.TempDir()
reportPath := filepath.Join(t.TempDir(), "report.json")
transcriptPath := writeFile(t, "transcript.json", `[ transcriptPath := writeFile(t, "transcript.json", `[
{"id":1,"speaker":"Alice","start":0.0,"end":1.0,"text":"Hello"} {"id":1,"speaker":"Alice","start":0.0,"end":1.0,"text":"Hello"}
]`) ]`)
@@ -2104,9 +2095,6 @@ func TestRunProcessProductionRegistryUnsupportedModuleFailsCleanly(t *testing.T)
"process", transcriptPath, "process", transcriptPath,
"--glossary", fixturePath("tiny_glossary.yaml"), "--glossary", fixturePath("tiny_glossary.yaml"),
"--modules", "made_up", "--modules", "made_up",
"--work-dir", workDir,
"--work-dir-retention", "always",
"--report-json", reportPath,
}, &stdout, &stderr) }, &stdout, &stderr)
if exitCode == 0 { if exitCode == 0 {
t.Fatal("expected failure exit code") t.Fatal("expected failure exit code")
@@ -2114,23 +2102,12 @@ func TestRunProcessProductionRegistryUnsupportedModuleFailsCleanly(t *testing.T)
if stdout.Len() != 0 { if stdout.Len() != 0 {
t.Fatalf("expected empty stdout on failure, got %q", stdout.String()) t.Fatalf("expected empty stdout on failure, got %q", stdout.String())
} }
if !strings.Contains(stderr.String(), "runner_execution") { if !strings.Contains(stderr.String(), "invalid CLI configuration") {
t.Fatalf("expected runner_execution failure on stderr, got %q", stderr.String()) t.Fatalf("expected config validation failure on stderr, got %q", stderr.String())
} }
if !strings.Contains(stderr.String(), "unsupported module key") { if !strings.Contains(stderr.String(), "unsupported module key") {
t.Fatalf("expected explicit unsupported module message, got %q", stderr.String()) t.Fatalf("expected explicit unsupported module message, got %q", stderr.String())
} }
report := readProcessReport(t, reportPath)
if report.Status != "failed" {
t.Fatalf("expected failed report status, got %q", report.Status)
}
if report.ErrorPhase != "runner_execution" {
t.Fatalf("expected runner_execution phase, got %q", report.ErrorPhase)
}
if !strings.Contains(report.ErrorMessage, "unsupported module key") {
t.Fatalf("expected report error message to mention unsupported module, got %q", report.ErrorMessage)
}
} }
func TestRunProcessExplicitUnsupportedModulesFailClearly(t *testing.T) { func TestRunProcessExplicitUnsupportedModulesFailClearly(t *testing.T) {
@@ -2384,7 +2361,7 @@ func TestRunProcessExplicitGrammarRejectedAndApplicationSkipAreDistinct(t *testi
} }
} }
func TestRunProcessExplicitGrammarMalformedLLMOutputFailsWithErrorLog(t *testing.T) { func TestRunProcessExplicitGrammarMalformedLLMOutputSucceedsWithWarning(t *testing.T) {
processProposalLLMClient = &fakeStructuredLLMClient{err: errors.New("malformed structured output")} processProposalLLMClient = &fakeStructuredLLMClient{err: errors.New("malformed structured output")}
t.Cleanup(func() { processProposalLLMClient = nil }) t.Cleanup(func() { processProposalLLMClient = nil })
@@ -2403,22 +2380,32 @@ func TestRunProcessExplicitGrammarMalformedLLMOutputFailsWithErrorLog(t *testing
"--work-dir-retention", "always", "--work-dir-retention", "always",
"--report-json", reportPath, "--report-json", reportPath,
}, &stdout, &stderr) }, &stdout, &stderr)
if exitCode == 0 { if exitCode != 0 {
t.Fatal("expected failure") t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
} }
if stdout.Len() != 0 { if stderr.Len() != 0 {
t.Fatalf("expected empty stdout on failure, got %q", stdout.String()) t.Fatalf("expected empty stderr on success, got %q", stderr.String())
} }
if !strings.Contains(stderr.String(), "runner_execution") { parsed, err := schema.ParseTranscriptJSON(stdout.Bytes())
t.Fatalf("expected runner_execution error, got %q", stderr.String()) if err != nil {
t.Fatalf("expected transcript stdout on success: %v", err)
}
if len(parsed.Segments) != 1 || parsed.Segments[0].Text != "hello" {
t.Fatalf("expected unchanged transcript, got %+v", parsed.Segments)
} }
runDir := onlyRunDir(t, workDir) runDir := onlyRunDir(t, workDir)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err != nil { if _, err := os.Stat(filepath.Join(runDir, "error.log")); err == nil || !os.IsNotExist(err) {
t.Fatalf("expected error.log on failed grammar run: %v", err) t.Fatalf("did not expect error.log on successful grammar run: %v", err)
} }
report := readProcessReport(t, reportPath) report := readProcessReport(t, reportPath)
if report.Status != "failed" || report.ErrorPhase != "runner_execution" { if report.Status != "success" || report.ErrorPhase != "" {
t.Fatalf("expected failed runner_execution report, got %+v", report) t.Fatalf("expected successful report, got %+v", report)
}
if len(report.ModuleResults) != 1 || len(report.ModuleResults[0].Warnings) != 1 {
t.Fatalf("expected one module warning, got %+v", report.ModuleResults)
}
if report.ModuleResults[0].Warnings[0].ReasonCode != "proposal_response_malformed" {
t.Fatalf("unexpected warning: %+v", report.ModuleResults[0].Warnings[0])
} }
} }
@@ -3041,7 +3028,7 @@ func TestRunProcessExplicitGlossaryRepeatedStagesUseDeterministicInstanceNamesAn
} }
} }
func TestRunProcessExplicitGlossaryMalformedLLMOutputFailsWithErrorLog(t *testing.T) { func TestRunProcessExplicitGlossaryMalformedLLMOutputSucceedsWithWarning(t *testing.T) {
processProposalLLMClient = &fakeStructuredLLMClient{err: errors.New("malformed structured output")} processProposalLLMClient = &fakeStructuredLLMClient{err: errors.New("malformed structured output")}
t.Cleanup(func() { processProposalLLMClient = nil }) t.Cleanup(func() { processProposalLLMClient = nil })
@@ -3060,22 +3047,29 @@ func TestRunProcessExplicitGlossaryMalformedLLMOutputFailsWithErrorLog(t *testin
"--work-dir-retention", "always", "--work-dir-retention", "always",
"--report-json", reportPath, "--report-json", reportPath,
}, &stdout, &stderr) }, &stdout, &stderr)
if exitCode == 0 { if exitCode != 0 {
t.Fatal("expected failure") t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
} }
if stdout.Len() != 0 { if stderr.Len() != 0 {
t.Fatalf("expected empty stdout on failure, got %q", stdout.String()) t.Fatalf("expected empty stderr on success, got %q", stderr.String())
} }
if !strings.Contains(stderr.String(), "runner_execution") { parsed, err := schema.ParseTranscriptJSON(stdout.Bytes())
t.Fatalf("expected runner_execution error, got %q", stderr.String()) if err != nil {
t.Fatalf("expected transcript stdout on success: %v", err)
}
if len(parsed.Segments) != 1 || parsed.Segments[0].Text != "hello" {
t.Fatalf("expected unchanged transcript, got %+v", parsed.Segments)
} }
runDir := onlyRunDir(t, workDir) runDir := onlyRunDir(t, workDir)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err != nil { if _, err := os.Stat(filepath.Join(runDir, "error.log")); err == nil || !os.IsNotExist(err) {
t.Fatalf("expected error.log on failed glossary run: %v", err) t.Fatalf("did not expect error.log on successful glossary run: %v", err)
} }
report := readProcessReport(t, reportPath) report := readProcessReport(t, reportPath)
if report.Status != "failed" || report.ErrorPhase != "runner_execution" { if report.Status != "success" || report.ErrorPhase != "" {
t.Fatalf("expected failed runner_execution report, got %+v", report) t.Fatalf("expected successful report, got %+v", report)
}
if len(report.ModuleResults) != 1 || len(report.ModuleResults[0].Warnings) != 1 {
t.Fatalf("expected one module warning, got %+v", report.ModuleResults)
} }
} }
@@ -3295,7 +3289,7 @@ func TestRunProcessExplicitHomophonesProtectedGlossaryTermRejected(t *testing.T)
} }
} }
func TestRunProcessExplicitHomophonesMalformedLLMOutputFailsWithErrorLog(t *testing.T) { func TestRunProcessExplicitHomophonesMalformedLLMOutputSucceedsWithWarning(t *testing.T) {
processProposalLLMClient = &fakeStructuredLLMClient{err: errors.New("malformed structured output")} processProposalLLMClient = &fakeStructuredLLMClient{err: errors.New("malformed structured output")}
t.Cleanup(func() { processProposalLLMClient = nil }) t.Cleanup(func() { processProposalLLMClient = nil })
@@ -3314,22 +3308,29 @@ func TestRunProcessExplicitHomophonesMalformedLLMOutputFailsWithErrorLog(t *test
"--work-dir-retention", "always", "--work-dir-retention", "always",
"--report-json", reportPath, "--report-json", reportPath,
}, &stdout, &stderr) }, &stdout, &stderr)
if exitCode == 0 { if exitCode != 0 {
t.Fatal("expected failure") t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
} }
if stdout.Len() != 0 { if stderr.Len() != 0 {
t.Fatalf("expected empty stdout on failure, got %q", stdout.String()) t.Fatalf("expected empty stderr on success, got %q", stderr.String())
} }
if !strings.Contains(stderr.String(), "runner_execution") { parsed, err := schema.ParseTranscriptJSON(stdout.Bytes())
t.Fatalf("expected runner_execution error, got %q", stderr.String()) if err != nil {
t.Fatalf("expected transcript stdout on success: %v", err)
}
if len(parsed.Segments) != 1 || parsed.Segments[0].Text != "hello" {
t.Fatalf("expected unchanged transcript, got %+v", parsed.Segments)
} }
runDir := onlyRunDir(t, workDir) runDir := onlyRunDir(t, workDir)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err != nil { if _, err := os.Stat(filepath.Join(runDir, "error.log")); err == nil || !os.IsNotExist(err) {
t.Fatalf("expected error.log on failed homophones run: %v", err) t.Fatalf("did not expect error.log on successful homophones run: %v", err)
} }
report := readProcessReport(t, reportPath) report := readProcessReport(t, reportPath)
if report.Status != "failed" || report.ErrorPhase != "runner_execution" { if report.Status != "success" || report.ErrorPhase != "" {
t.Fatalf("expected failed runner_execution report, got %+v", report) t.Fatalf("expected successful report, got %+v", report)
}
if len(report.ModuleResults) != 1 || len(report.ModuleResults[0].Warnings) != 1 {
t.Fatalf("expected one module warning, got %+v", report.ModuleResults)
} }
} }
@@ -3674,7 +3675,7 @@ func TestRunProcessExplicitSpokenWordProtectedGlossaryTermRejected(t *testing.T)
} }
} }
func TestRunProcessExplicitSpokenWordMalformedLLMOutputFailsWithErrorLog(t *testing.T) { func TestRunProcessExplicitSpokenWordMalformedLLMOutputSucceedsWithWarning(t *testing.T) {
processProposalLLMClient = &fakeStructuredLLMClient{err: errors.New("malformed structured output")} processProposalLLMClient = &fakeStructuredLLMClient{err: errors.New("malformed structured output")}
t.Cleanup(func() { processProposalLLMClient = nil }) t.Cleanup(func() { processProposalLLMClient = nil })
@@ -3693,22 +3694,29 @@ func TestRunProcessExplicitSpokenWordMalformedLLMOutputFailsWithErrorLog(t *test
"--work-dir-retention", "always", "--work-dir-retention", "always",
"--report-json", reportPath, "--report-json", reportPath,
}, &stdout, &stderr) }, &stdout, &stderr)
if exitCode == 0 { if exitCode != 0 {
t.Fatal("expected failure") t.Fatalf("expected success, got %d stderr=%q", exitCode, stderr.String())
} }
if stdout.Len() != 0 { if stderr.Len() != 0 {
t.Fatalf("expected empty stdout on failure, got %q", stdout.String()) t.Fatalf("expected empty stderr on success, got %q", stderr.String())
} }
if !strings.Contains(stderr.String(), "runner_execution") { parsed, err := schema.ParseTranscriptJSON(stdout.Bytes())
t.Fatalf("expected runner_execution error, got %q", stderr.String()) if err != nil {
t.Fatalf("expected transcript stdout on success: %v", err)
}
if len(parsed.Segments) != 1 || parsed.Segments[0].Text != "hello" {
t.Fatalf("expected unchanged transcript, got %+v", parsed.Segments)
} }
runDir := onlyRunDir(t, workDir) runDir := onlyRunDir(t, workDir)
if _, err := os.Stat(filepath.Join(runDir, "error.log")); err != nil { if _, err := os.Stat(filepath.Join(runDir, "error.log")); err == nil || !os.IsNotExist(err) {
t.Fatalf("expected error.log on failed spoken_word run: %v", err) t.Fatalf("did not expect error.log on successful spoken_word run: %v", err)
} }
report := readProcessReport(t, reportPath) report := readProcessReport(t, reportPath)
if report.Status != "failed" || report.ErrorPhase != "runner_execution" { if report.Status != "success" || report.ErrorPhase != "" {
t.Fatalf("expected failed runner_execution report, got %+v", report) t.Fatalf("expected successful report, got %+v", report)
}
if len(report.ModuleResults) != 1 || len(report.ModuleResults[0].Warnings) != 1 {
t.Fatalf("expected one module warning, got %+v", report.ModuleResults)
} }
} }
@@ -4328,12 +4336,7 @@ func writeFile(t *testing.T, name string, content string) string {
} }
func readFile(t *testing.T, path string) []byte { func readFile(t *testing.T, path string) []byte {
t.Helper() return testsupport.ReadFile(t, path)
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("failed to read file %q: %v", path, err)
}
return data
} }
func readProcessReport(t *testing.T, path string) reporting.ProcessReport { func readProcessReport(t *testing.T, path string) reporting.ProcessReport {
@@ -4347,13 +4350,5 @@ func readProcessReport(t *testing.T, path string) reporting.ProcessReport {
} }
func onlyRunDir(t *testing.T, workDir string) string { func onlyRunDir(t *testing.T, workDir string) string {
t.Helper() return testsupport.OnlyRunDir(t, workDir)
entries, err := os.ReadDir(workDir)
if err != nil {
t.Fatalf("failed to read work dir %q: %v", workDir, err)
}
if len(entries) != 1 {
t.Fatalf("expected exactly one run dir in %q, got %d", workDir, len(entries))
}
return filepath.Join(workDir, entries[0].Name())
} }

View File

@@ -78,25 +78,22 @@ func (c *subprocessTestLLMClient) CompleteStructured(ctx context.Context, req co
} }
} }
case "mid_pipeline_fail": case "mid_pipeline_fail":
switch target := out.(type) { if _, ok := out.(*proposal_generation.StructuredCorrectionSet); ok {
case *proposal_generation.StructuredCorrectionSet:
c.mu.Lock() c.mu.Lock()
c.proposals++ c.proposals++
proposalCall := c.proposals proposalCall := c.proposals
c.mu.Unlock() c.mu.Unlock()
if proposalCall >= 3 { if proposalCall >= 3 {
*target = proposal_generation.StructuredCorrectionSet{ return contracts.StructuredCompletionResponse{}, errors.New("synthetic mid-pipeline failure")
Corrections: []proposal_generation.StructuredCorrectionProposal{ }
{TargetSegmentID: 0, OriginalText: "x", CorrectedText: "y", Confidence: 0.99}, }
}, switch target := out.(type) {
} case *proposal_generation.StructuredCorrectionSet:
} else { *target = proposal_generation.StructuredCorrectionSet{
*target = proposal_generation.StructuredCorrectionSet{ Corrections: []proposal_generation.StructuredCorrectionProposal{
Corrections: []proposal_generation.StructuredCorrectionProposal{ {TargetSegmentID: 1, OriginalText: "Segment", CorrectedText: "Segment", Confidence: 0.99},
{TargetSegmentID: 1, OriginalText: "Segment", CorrectedText: "Segment", Confidence: 0.99}, },
},
}
} }
case *validators.LLMValidationResponse: case *validators.LLMValidationResponse:
*target = validators.LLMValidationResponse{Validations: nil} *target = validators.LLMValidationResponse{Validations: nil}

View File

@@ -3,6 +3,8 @@ package config
import ( import (
"fmt" "fmt"
"strings" "strings"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
) )
type WorkDirRetention string type WorkDirRetention string
@@ -14,7 +16,7 @@ const (
) )
const ( const (
DefaultModulesCSV = "glossary,homophones,glossary,spoken_word,grammar" DefaultModulesCSV = modulecatalog.KeyGlossary + "," + modulecatalog.KeyHomophones + "," + modulecatalog.KeyGlossary + "," + modulecatalog.KeySpokenWord + "," + modulecatalog.KeyGrammar
DefaultOutputSchema = "bare-segments" DefaultOutputSchema = "bare-segments"
DefaultPrimaryModel = "openrouter/google/gemma-4-31b-it" DefaultPrimaryModel = "openrouter/google/gemma-4-31b-it"
DefaultPrimaryBaseURL = "https://openrouter.ai/api/v1" DefaultPrimaryBaseURL = "https://openrouter.ai/api/v1"

View File

@@ -4,6 +4,9 @@ import (
"reflect" "reflect"
"strings" "strings"
"testing" "testing"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
"gitea.maximumdirect.net/eric/audita/internal/core/outputschema"
) )
func TestDefaultConfigValues(t *testing.T) { func TestDefaultConfigValues(t *testing.T) {
@@ -318,6 +321,38 @@ func TestValidationFailures(t *testing.T) {
} }
} }
func TestValidationRejectsUnsupportedModuleKey(t *testing.T) {
cfg := Default()
cfg.Modules = []string{modulecatalog.KeyGlossary, "made_up"}
err := cfg.Validate()
if err == nil {
t.Fatalf("expected validation error for unsupported module key")
}
if !strings.Contains(err.Error(), `unsupported module key "made_up"`) {
t.Fatalf("expected unsupported module key error, got %q", err.Error())
}
}
func TestValidationAllowsRepeatedSupportedModuleKeys(t *testing.T) {
cfg := Default()
cfg.Modules = []string{modulecatalog.KeyGlossary, modulecatalog.KeyGlossary, modulecatalog.KeyGrammar}
if err := cfg.Validate(); err != nil {
t.Fatalf("expected repeated supported module keys to validate, got %v", err)
}
}
func TestValidationAcceptsAllSupportedOutputSchemas(t *testing.T) {
for _, schemaKey := range outputschema.SupportedKeys() {
cfg := Default()
cfg.OutputSchema = schemaKey
if err := cfg.Validate(); err != nil {
t.Fatalf("expected output schema %q to validate, got %v", schemaKey, err)
}
}
}
func TestEffectiveValidationLLMInheritance(t *testing.T) { func TestEffectiveValidationLLMInheritance(t *testing.T) {
cfg := Default() cfg := Default()
cfg.PrimaryLLM.APIKey = "primary-key" cfg.PrimaryLLM.APIKey = "primary-key"

View File

@@ -0,0 +1,119 @@
package config
import (
"fmt"
"os"
"strings"
)
type EffectiveConfigErrorKind string
const (
EffectiveConfigErrorResolvePath EffectiveConfigErrorKind = "resolve_path"
EffectiveConfigErrorLoadFile EffectiveConfigErrorKind = "load_file"
EffectiveConfigErrorApplyFile EffectiveConfigErrorKind = "apply_file"
EffectiveConfigErrorApplyEnv EffectiveConfigErrorKind = "apply_env"
)
type EffectiveConfigError struct {
Kind EffectiveConfigErrorKind
Err error
}
func (e *EffectiveConfigError) Error() string {
if e == nil || e.Err == nil {
return ""
}
return e.Err.Error()
}
func (e *EffectiveConfigError) Unwrap() error {
if e == nil {
return nil
}
return e.Err
}
type EffectiveConfig struct {
Config Config
ConfigPath string
ConfigSource string
ConfigVersion *int
}
func ResolveConfigPath(cliConfigPath string, cliConfigPathSet bool) (path string, source string, err error) {
return resolveConfigPathWithLookup(cliConfigPath, cliConfigPathSet, os.LookupEnv, os.Stat, DefaultConfigSearchPaths)
}
func LoadEffectiveConfig(cliConfigPath string, cliConfigPathSet bool) (EffectiveConfig, error) {
return loadEffectiveConfigWithLookup(cliConfigPath, cliConfigPathSet, os.LookupEnv, os.Stat, DefaultConfigSearchPaths)
}
func loadEffectiveConfigWithLookup(cliConfigPath string, cliConfigPathSet bool, lookup func(string) (string, bool), statPath func(string) (os.FileInfo, error), defaultSearchPaths []string) (EffectiveConfig, error) {
configPath, configSource, err := resolveConfigPathWithLookup(cliConfigPath, cliConfigPathSet, lookup, statPath, defaultSearchPaths)
if err != nil {
return EffectiveConfig{}, &EffectiveConfigError{Kind: EffectiveConfigErrorResolvePath, Err: err}
}
cfg := Default()
var configVersion *int
if configPath != "" {
fileCfg, fileErr := LoadFileConfig(configPath)
if fileErr != nil {
return EffectiveConfig{}, &EffectiveConfigError{Kind: EffectiveConfigErrorLoadFile, Err: fileErr}
}
if applyErr := cfg.ApplyFileConfig(fileCfg); applyErr != nil {
return EffectiveConfig{}, &EffectiveConfigError{Kind: EffectiveConfigErrorApplyFile, Err: applyErr}
}
configVersion = &fileCfg.Version
}
if applyEnvErr := cfg.applyEnvOverrides(lookup); applyEnvErr != nil {
return EffectiveConfig{}, &EffectiveConfigError{Kind: EffectiveConfigErrorApplyEnv, Err: applyEnvErr}
}
return EffectiveConfig{
Config: cfg,
ConfigPath: configPath,
ConfigSource: configSource,
ConfigVersion: configVersion,
}, nil
}
func resolveConfigPathWithLookup(cliConfigPath string, cliConfigPathSet bool, lookup func(string) (string, bool), statPath func(string) (os.FileInfo, error), defaultSearchPaths []string) (path string, source string, err error) {
if cliConfigPathSet {
path = strings.TrimSpace(cliConfigPath)
if path == "" {
return "", "", fmt.Errorf("--config requires a non-empty path")
}
if _, statErr := statPath(path); statErr != nil {
if os.IsNotExist(statErr) {
return "", "", fmt.Errorf("config file not found: %s", path)
}
return "", "", fmt.Errorf("cannot access config file %s: %w", path, statErr)
}
return path, "flag", nil
}
if raw, ok := lookup("AUDITA_CONFIG"); ok {
path = strings.TrimSpace(raw)
if path == "" {
return "", "", fmt.Errorf("AUDITA_CONFIG must not be empty")
}
if _, statErr := statPath(path); statErr != nil {
if os.IsNotExist(statErr) {
return "", "", fmt.Errorf("config file not found: %s", path)
}
return "", "", fmt.Errorf("cannot access config file %s: %w", path, statErr)
}
return path, "env", nil
}
for _, defaultPath := range defaultSearchPaths {
if _, statErr := statPath(defaultPath); statErr == nil {
return defaultPath, "default", nil
} else if !os.IsNotExist(statErr) {
return "", "", fmt.Errorf("cannot access config file %s: %w", defaultPath, statErr)
}
}
return "", "", nil
}

View File

@@ -0,0 +1,163 @@
package config
import (
"os"
"path/filepath"
"strings"
"testing"
)
func TestResolveConfigPathWithLookupMatrix(t *testing.T) {
statFor := func(existing map[string]bool) func(string) (os.FileInfo, error) {
return func(path string) (os.FileInfo, error) {
if existing[path] {
return nil, nil
}
return nil, os.ErrNotExist
}
}
tests := []struct {
name string
cliPath string
cliPathSet bool
lookup func(string) (string, bool)
stat func(string) (os.FileInfo, error)
defaultSearchPaths []string
wantPath string
wantSource string
wantErrContains string
}{
{
name: "explicit config path",
cliPath: "/tmp/explicit.yml",
cliPathSet: true,
lookup: func(string) (string, bool) { return "", false },
stat: statFor(map[string]bool{"/tmp/explicit.yml": true}),
defaultSearchPaths: []string{
"/usr/local/etc/audita/config.yml",
"/etc/audita/config.yml",
},
wantPath: "/tmp/explicit.yml",
wantSource: "flag",
},
{
name: "env config path",
cliPathSet: false,
lookup: func(key string) (string, bool) {
if key == "AUDITA_CONFIG" {
return "/tmp/from-env.yml", true
}
return "", false
},
stat: statFor(map[string]bool{"/tmp/from-env.yml": true}),
defaultSearchPaths: []string{"/usr/local/etc/audita/config.yml", "/etc/audita/config.yml"},
wantPath: "/tmp/from-env.yml",
wantSource: "env",
},
{
name: "default search path",
cliPathSet: false,
lookup: func(string) (string, bool) { return "", false },
stat: statFor(map[string]bool{
"/usr/local/etc/audita/config.yml": true,
"/etc/audita/config.yml": true,
}),
defaultSearchPaths: []string{"/usr/local/etc/audita/config.yml", "/etc/audita/config.yml"},
wantPath: "/usr/local/etc/audita/config.yml",
wantSource: "default",
},
{
name: "explicit missing path",
cliPath: "/tmp/missing.yml",
cliPathSet: true,
lookup: func(string) (string, bool) { return "", false },
stat: statFor(map[string]bool{}),
defaultSearchPaths: []string{
"/usr/local/etc/audita/config.yml",
"/etc/audita/config.yml",
},
wantErrContains: "config file not found",
},
{
name: "missing env path",
cliPathSet: false,
lookup: func(key string) (string, bool) {
if key == "AUDITA_CONFIG" {
return "/tmp/missing-from-env.yml", true
}
return "", false
},
stat: statFor(map[string]bool{}),
defaultSearchPaths: []string{"/usr/local/etc/audita/config.yml", "/etc/audita/config.yml"},
wantErrContains: "config file not found",
},
{
name: "missing default paths",
cliPathSet: false,
lookup: func(string) (string, bool) { return "", false },
stat: statFor(map[string]bool{}),
defaultSearchPaths: []string{
"/usr/local/etc/audita/config.yml",
"/etc/audita/config.yml",
},
wantPath: "",
wantSource: "",
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
gotPath, gotSource, err := resolveConfigPathWithLookup(tc.cliPath, tc.cliPathSet, tc.lookup, tc.stat, tc.defaultSearchPaths)
if tc.wantErrContains != "" {
if err == nil || !strings.Contains(err.Error(), tc.wantErrContains) {
t.Fatalf("expected error containing %q, got %v", tc.wantErrContains, err)
}
return
}
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if gotPath != tc.wantPath || gotSource != tc.wantSource {
t.Fatalf("unexpected result: got path=%q source=%q, want path=%q source=%q", gotPath, gotSource, tc.wantPath, tc.wantSource)
}
})
}
}
func TestLoadEffectiveConfigWithLookupAppliesDefaultsFileThenEnv(t *testing.T) {
tempDir := t.TempDir()
configPath := filepath.Join(tempDir, "config.yml")
configYAML := "version: 1\nllm:\n proposal:\n model: file-model\n"
if err := os.WriteFile(configPath, []byte(configYAML), 0o644); err != nil {
t.Fatalf("write config file: %v", err)
}
lookup := func(key string) (string, bool) {
switch key {
case "AUDITA_CONFIG":
return configPath, true
case "AUDITA_MODEL":
return "env-model", true
default:
return "", false
}
}
result, err := loadEffectiveConfigWithLookup("", false, lookup, os.Stat, DefaultConfigSearchPaths)
if err != nil {
t.Fatalf("loadEffectiveConfigWithLookup error: %v", err)
}
if result.ConfigPath != configPath {
t.Fatalf("unexpected config path: %q", result.ConfigPath)
}
if result.ConfigSource != "env" {
t.Fatalf("unexpected config source: %q", result.ConfigSource)
}
if result.ConfigVersion == nil || *result.ConfigVersion != SupportedFileConfigVersion {
t.Fatalf("unexpected config version: %#v", result.ConfigVersion)
}
if result.Config.PrimaryLLM.Model != "env-model" {
t.Fatalf("expected env override to win over file value, got %q", result.Config.PrimaryLLM.Model)
}
}

View File

@@ -3,6 +3,9 @@ package config
import ( import (
"fmt" "fmt"
"strings" "strings"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
"gitea.maximumdirect.net/eric/audita/internal/core/outputschema"
) )
func (c Config) Validate() error { func (c Config) Validate() error {
@@ -12,19 +15,19 @@ func (c Config) Validate() error {
issues = append(issues, "modules must not be empty") issues = append(issues, "modules must not be empty")
} }
for _, module := range c.Modules { for _, module := range c.Modules {
if strings.TrimSpace(module) == "" { moduleKey := strings.TrimSpace(module)
if moduleKey == "" {
issues = append(issues, "modules must not contain empty values") issues = append(issues, "modules must not contain empty values")
break break
} }
if !modulecatalog.IsSupported(moduleKey) {
issues = append(issues, fmt.Sprintf("unsupported module key %q", moduleKey))
}
} }
if strings.TrimSpace(c.OutputSchema) == "" { if strings.TrimSpace(c.OutputSchema) == "" {
issues = append(issues, "output schema must not be empty") issues = append(issues, "output schema must not be empty")
} else { } else if !outputschema.IsSupported(c.OutputSchema) {
switch strings.TrimSpace(c.OutputSchema) { issues = append(issues, fmt.Sprintf("unsupported output schema %q", c.OutputSchema))
case "bare-segments", "audita-v1":
default:
issues = append(issues, fmt.Sprintf("unsupported output schema %q", c.OutputSchema))
}
} }
if c.PrimaryLLM.TimeoutSeconds <= 0 { if c.PrimaryLLM.TimeoutSeconds <= 0 {

View File

@@ -0,0 +1,42 @@
package diagnostics
import (
"path/filepath"
"gitea.maximumdirect.net/eric/audita/internal/core/reporting"
)
const (
ArtifactSourceTranscript = "source-transcript.json"
ArtifactParsedSourceTranscript = "source-transcript-parsed.json"
ArtifactNormalizedTranscript = "normalized-transcript.json"
ArtifactNormalizationSummary = "normalization-summary.json"
ArtifactChunkingSummary = "chunking-summary.json"
ArtifactUtilizationSummary = "utilization-diagnostics.json"
ArtifactCorrectionLedger = "correction-ledger.json"
ArtifactInvocationMetadata = "invocation.json"
ArtifactEffectiveConfig = "effective-config.json"
ArtifactReport = "report.json"
ArtifactErrorLog = "error.log"
)
func BuildDiagnosticsMetadata(runDirectoryPath string, runSucceeded bool) reporting.DiagnosticsMetadata {
metadata := reporting.DiagnosticsMetadata{
DirectoryPath: runDirectoryPath,
SourceTranscriptPath: filepath.Join(runDirectoryPath, ArtifactSourceTranscript),
ParsedSourceTranscriptPath: filepath.Join(runDirectoryPath, ArtifactParsedSourceTranscript),
NormalizedTranscriptPath: filepath.Join(runDirectoryPath, ArtifactNormalizedTranscript),
NormalizationSummaryPath: filepath.Join(runDirectoryPath, ArtifactNormalizationSummary),
ChunkingSummaryPath: filepath.Join(runDirectoryPath, ArtifactChunkingSummary),
UtilizationSummaryPath: filepath.Join(runDirectoryPath, ArtifactUtilizationSummary),
CorrectionLedgerPath: filepath.Join(runDirectoryPath, ArtifactCorrectionLedger),
InvocationMetadataPath: filepath.Join(runDirectoryPath, ArtifactInvocationMetadata),
RedactedEffectiveConfigPath: filepath.Join(runDirectoryPath, ArtifactEffectiveConfig),
}
if !runSucceeded {
metadata.ErrorLogPath = filepath.Join(runDirectoryPath, ArtifactErrorLog)
}
return metadata
}

View File

@@ -0,0 +1,55 @@
package diagnostics
import (
"path/filepath"
"testing"
)
func TestBuildDiagnosticsMetadataSuccessPathsMatchArtifactConstants(t *testing.T) {
runPath := filepath.Join("tmp", "run-123")
metadata := BuildDiagnosticsMetadata(runPath, true)
if metadata.DirectoryPath != runPath {
t.Fatalf("unexpected diagnostics directory path: got=%q want=%q", metadata.DirectoryPath, runPath)
}
if metadata.SourceTranscriptPath != filepath.Join(runPath, ArtifactSourceTranscript) {
t.Fatalf("unexpected source transcript path: %q", metadata.SourceTranscriptPath)
}
if metadata.ParsedSourceTranscriptPath != filepath.Join(runPath, ArtifactParsedSourceTranscript) {
t.Fatalf("unexpected parsed source transcript path: %q", metadata.ParsedSourceTranscriptPath)
}
if metadata.NormalizedTranscriptPath != filepath.Join(runPath, ArtifactNormalizedTranscript) {
t.Fatalf("unexpected normalized transcript path: %q", metadata.NormalizedTranscriptPath)
}
if metadata.NormalizationSummaryPath != filepath.Join(runPath, ArtifactNormalizationSummary) {
t.Fatalf("unexpected normalization summary path: %q", metadata.NormalizationSummaryPath)
}
if metadata.ChunkingSummaryPath != filepath.Join(runPath, ArtifactChunkingSummary) {
t.Fatalf("unexpected chunking summary path: %q", metadata.ChunkingSummaryPath)
}
if metadata.UtilizationSummaryPath != filepath.Join(runPath, ArtifactUtilizationSummary) {
t.Fatalf("unexpected utilization summary path: %q", metadata.UtilizationSummaryPath)
}
if metadata.CorrectionLedgerPath != filepath.Join(runPath, ArtifactCorrectionLedger) {
t.Fatalf("unexpected correction ledger path: %q", metadata.CorrectionLedgerPath)
}
if metadata.InvocationMetadataPath != filepath.Join(runPath, ArtifactInvocationMetadata) {
t.Fatalf("unexpected invocation metadata path: %q", metadata.InvocationMetadataPath)
}
if metadata.RedactedEffectiveConfigPath != filepath.Join(runPath, ArtifactEffectiveConfig) {
t.Fatalf("unexpected redacted effective config path: %q", metadata.RedactedEffectiveConfigPath)
}
if metadata.ErrorLogPath != "" {
t.Fatalf("did not expect error log path on success: %q", metadata.ErrorLogPath)
}
}
func TestBuildDiagnosticsMetadataFailureIncludesErrorLogPath(t *testing.T) {
runPath := filepath.Join("tmp", "run-123")
metadata := BuildDiagnosticsMetadata(runPath, false)
want := filepath.Join(runPath, ArtifactErrorLog)
if metadata.ErrorLogPath != want {
t.Fatalf("unexpected error log path: got=%q want=%q", metadata.ErrorLogPath, want)
}
}

View File

@@ -106,7 +106,7 @@ func (r *RunDirectory) WriteInvocationMetadata(metadata InvocationMetadata) erro
metadata.StartedAt = r.createdAt metadata.StartedAt = r.createdAt
} }
path := filepath.Join(r.path, "invocation.json") path := filepath.Join(r.path, ArtifactInvocationMetadata)
bytes, err := json.MarshalIndent(metadata, "", " ") bytes, err := json.MarshalIndent(metadata, "", " ")
if err != nil { if err != nil {
return fmt.Errorf("failed to marshal invocation metadata: %w", err) return fmt.Errorf("failed to marshal invocation metadata: %w", err)
@@ -120,7 +120,7 @@ func (r *RunDirectory) WriteInvocationMetadata(metadata InvocationMetadata) erro
// WriteEffectiveConfig writes redacted effective config metadata for this run. // WriteEffectiveConfig writes redacted effective config metadata for this run.
func (r *RunDirectory) WriteEffectiveConfig(cfg config.Config) error { func (r *RunDirectory) WriteEffectiveConfig(cfg config.Config) error {
path := filepath.Join(r.path, "effective-config.json") path := filepath.Join(r.path, ArtifactEffectiveConfig)
redacted := cfg.Redacted() redacted := cfg.Redacted()
bytes, err := json.MarshalIndent(redacted, "", " ") bytes, err := json.MarshalIndent(redacted, "", " ")
if err != nil { if err != nil {
@@ -136,13 +136,13 @@ func (r *RunDirectory) WriteEffectiveConfig(cfg config.Config) error {
// WriteSourceTranscript writes the source transcript artifact // WriteSourceTranscript writes the source transcript artifact
func (r *RunDirectory) WriteSourceTranscript(transcript *schema.SourceTranscript, raw []byte) error { func (r *RunDirectory) WriteSourceTranscript(transcript *schema.SourceTranscript, raw []byte) error {
// Write raw source for reference // Write raw source for reference
sourcePath := filepath.Join(r.path, "source-transcript.json") sourcePath := filepath.Join(r.path, ArtifactSourceTranscript)
if err := os.WriteFile(sourcePath, raw, 0o644); err != nil { if err := os.WriteFile(sourcePath, raw, 0o644); err != nil {
return fmt.Errorf("failed to write source transcript: %w", err) return fmt.Errorf("failed to write source transcript: %w", err)
} }
// Write parsed source for debugging // Write parsed source for debugging
parsedPath := filepath.Join(r.path, "source-transcript-parsed.json") parsedPath := filepath.Join(r.path, ArtifactParsedSourceTranscript)
parsedBytes, err := json.MarshalIndent(transcript, "", " ") parsedBytes, err := json.MarshalIndent(transcript, "", " ")
if err != nil { if err != nil {
return fmt.Errorf("failed to marshal parsed source transcript: %w", err) return fmt.Errorf("failed to marshal parsed source transcript: %w", err)
@@ -157,7 +157,7 @@ func (r *RunDirectory) WriteSourceTranscript(transcript *schema.SourceTranscript
// WriteNormalizedTranscript writes the normalized transcript artifact // WriteNormalizedTranscript writes the normalized transcript artifact
func (r *RunDirectory) WriteNormalizedTranscript(transcript *schema.Transcript) error { func (r *RunDirectory) WriteNormalizedTranscript(transcript *schema.Transcript) error {
normalizedPath := filepath.Join(r.path, "normalized-transcript.json") normalizedPath := filepath.Join(r.path, ArtifactNormalizedTranscript)
bytes, err := schema.TranscriptToJSON(transcript) bytes, err := schema.TranscriptToJSON(transcript)
if err != nil { if err != nil {
return fmt.Errorf("failed to serialize normalized transcript: %w", err) return fmt.Errorf("failed to serialize normalized transcript: %w", err)
@@ -170,7 +170,7 @@ func (r *RunDirectory) WriteNormalizedTranscript(transcript *schema.Transcript)
// WriteNormalizationSummary writes the normalization summary artifact // WriteNormalizationSummary writes the normalization summary artifact
func (r *RunDirectory) WriteNormalizationSummary(summary *normalization.NormalizationSummary) error { func (r *RunDirectory) WriteNormalizationSummary(summary *normalization.NormalizationSummary) error {
summaryPath := filepath.Join(r.path, "normalization-summary.json") summaryPath := filepath.Join(r.path, ArtifactNormalizationSummary)
bytes, err := json.MarshalIndent(summary, "", " ") bytes, err := json.MarshalIndent(summary, "", " ")
if err != nil { if err != nil {
return fmt.Errorf("failed to marshal normalization summary: %w", err) return fmt.Errorf("failed to marshal normalization summary: %w", err)
@@ -184,7 +184,7 @@ func (r *RunDirectory) WriteNormalizationSummary(summary *normalization.Normaliz
// WriteReport writes the authoritative report artifact // WriteReport writes the authoritative report artifact
func (r *RunDirectory) WriteReport(report reporting.ProcessReport) error { func (r *RunDirectory) WriteReport(report reporting.ProcessReport) error {
reportPath := filepath.Join(r.path, "report.json") reportPath := filepath.Join(r.path, ArtifactReport)
bytes, err := json.MarshalIndent(report, "", " ") bytes, err := json.MarshalIndent(report, "", " ")
if err != nil { if err != nil {
return fmt.Errorf("failed to marshal report: %w", err) return fmt.Errorf("failed to marshal report: %w", err)
@@ -198,13 +198,13 @@ func (r *RunDirectory) WriteReport(report reporting.ProcessReport) error {
// WriteErrorLog writes an error log on failure // WriteErrorLog writes an error log on failure
func (r *RunDirectory) WriteErrorLog(errorMessage string) error { func (r *RunDirectory) WriteErrorLog(errorMessage string) error {
errorPath := filepath.Join(r.path, "error.log") errorPath := filepath.Join(r.path, ArtifactErrorLog)
return os.WriteFile(errorPath, []byte(errorMessage+"\n"), 0o644) return os.WriteFile(errorPath, []byte(errorMessage+"\n"), 0o644)
} }
// WriteChunkingSummary writes the chunking summary artifact // WriteChunkingSummary writes the chunking summary artifact
func (r *RunDirectory) WriteChunkingSummary(summary *chunking.DetailedSummary) error { func (r *RunDirectory) WriteChunkingSummary(summary *chunking.DetailedSummary) error {
summaryPath := filepath.Join(r.path, "chunking-summary.json") summaryPath := filepath.Join(r.path, ArtifactChunkingSummary)
bytes, err := json.MarshalIndent(summary, "", " ") bytes, err := json.MarshalIndent(summary, "", " ")
if err != nil { if err != nil {
return fmt.Errorf("failed to marshal chunking summary: %w", err) return fmt.Errorf("failed to marshal chunking summary: %w", err)

View File

@@ -0,0 +1,35 @@
package modulecatalog
import "strings"
const (
KeyGlossary = "glossary"
KeyHomophones = "homophones"
KeySpokenWord = "spoken_word"
KeyGrammar = "grammar"
)
var supportedKeys = []string{
KeyGlossary,
KeyHomophones,
KeySpokenWord,
KeyGrammar,
}
var supportedKeySet = map[string]struct{}{
KeyGlossary: {},
KeyHomophones: {},
KeySpokenWord: {},
KeyGrammar: {},
}
func SupportedKeys() []string {
out := make([]string, len(supportedKeys))
copy(out, supportedKeys)
return out
}
func IsSupported(key string) bool {
_, ok := supportedKeySet[strings.TrimSpace(key)]
return ok
}

View File

@@ -0,0 +1,24 @@
package modulecatalog
import (
"reflect"
"testing"
)
func TestSupportedKeys(t *testing.T) {
want := []string{KeyGlossary, KeyHomophones, KeySpokenWord, KeyGrammar}
if got := SupportedKeys(); !reflect.DeepEqual(got, want) {
t.Fatalf("unexpected supported keys: got=%v want=%v", got, want)
}
}
func TestIsSupported(t *testing.T) {
for _, key := range SupportedKeys() {
if !IsSupported(key) {
t.Fatalf("expected key %q to be supported", key)
}
}
if IsSupported("made_up") {
t.Fatalf("did not expect made_up to be supported")
}
}

View File

@@ -31,15 +31,31 @@ var definitions = map[string]Definition{
}, },
} }
var supportedKeys = []string{
SchemaBareSegments,
SchemaAuditaV1,
}
func SupportedKeys() []string {
out := make([]string, len(supportedKeys))
copy(out, supportedKeys)
return out
}
func IsSupported(key string) bool {
_, ok := definitions[strings.TrimSpace(key)]
return ok
}
func Resolve(key string) (Definition, error) { func Resolve(key string) (Definition, error) {
normalized := strings.TrimSpace(key) normalized := strings.TrimSpace(key)
if normalized == "" { if normalized == "" {
return Definition{}, fmt.Errorf("output schema must not be empty") return Definition{}, fmt.Errorf("output schema must not be empty")
} }
def, ok := definitions[normalized] if !IsSupported(normalized) {
if !ok {
return Definition{}, fmt.Errorf("unsupported output schema %q", normalized) return Definition{}, fmt.Errorf("unsupported output schema %q", normalized)
} }
def := definitions[normalized]
return def, nil return def, nil
} }

View File

@@ -2,6 +2,7 @@ package outputschema
import ( import (
"encoding/json" "encoding/json"
"reflect"
"strings" "strings"
"testing" "testing"
@@ -56,3 +57,19 @@ func TestResolveUnknown(t *testing.T) {
t.Fatalf("expected unsupported output schema error, got %v", err) t.Fatalf("expected unsupported output schema error, got %v", err)
} }
} }
func TestSupportedKeysAndIsSupported(t *testing.T) {
want := []string{SchemaBareSegments, SchemaAuditaV1}
if got := SupportedKeys(); !reflect.DeepEqual(got, want) {
t.Fatalf("unexpected supported schema keys: got=%v want=%v", got, want)
}
for _, key := range want {
if !IsSupported(key) {
t.Fatalf("expected schema key %q to be supported", key)
}
}
if IsSupported("seriatim-intermediate") {
t.Fatalf("did not expect unsupported schema to be reported as supported")
}
}

View File

@@ -7,6 +7,7 @@ import (
"time" "time"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals" "gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
) )
type ProcessReport struct { type ProcessReport struct {
@@ -46,18 +47,19 @@ type ReportMetadata struct {
} }
type ModuleReport struct { type ModuleReport struct {
ModuleKey string `json:"module_key"` ModuleKey string `json:"module_key"`
ModuleInstance string `json:"module_instance"` ModuleInstance string `json:"module_instance"`
ReplacementPolicy string `json:"replacement_policy,omitempty"` ReplacementPolicy string `json:"replacement_policy,omitempty"`
Status string `json:"status"` Status string `json:"status"`
ProposalCount int `json:"proposal_count"` ProposalCount int `json:"proposal_count"`
ValidatorDecisions []ValidatorDecisionReport `json:"validator_decisions,omitempty"` Warnings []stagewarnings.StageWarning `json:"warnings,omitempty"`
ValidatorRejected []ValidatorRejectedReport `json:"validator_rejected,omitempty"` ValidatorDecisions []ValidatorDecisionReport `json:"validator_decisions,omitempty"`
AppliedChanges []proposals.AppliedChange `json:"applied_changes,omitempty"` ValidatorRejected []ValidatorRejectedReport `json:"validator_rejected,omitempty"`
SkippedChanges []proposals.SkippedChange `json:"skipped_changes,omitempty"` AppliedChanges []proposals.AppliedChange `json:"applied_changes,omitempty"`
ErrorMessage string `json:"error_message,omitempty"` SkippedChanges []proposals.SkippedChange `json:"skipped_changes,omitempty"`
StartedAt *time.Time `json:"started_at,omitempty"` ErrorMessage string `json:"error_message,omitempty"`
CompletedAt *time.Time `json:"completed_at,omitempty"` StartedAt *time.Time `json:"started_at,omitempty"`
CompletedAt *time.Time `json:"completed_at,omitempty"`
} }
type ValidatorDecisionReport struct { type ValidatorDecisionReport struct {

View File

@@ -47,7 +47,7 @@ func (h testChunkProposalHarness) collectEnrichedProposals(
return nil, err return nil, err
} }
for _, proposal := range base { for _, proposal := range base.Proposals {
sectionIndex := section.Index sectionIndex := section.Index
out = append(out, proposals.EnrichedCorrectionProposal{ out = append(out, proposals.EnrichedCorrectionProposal{
CorrectionProposal: proposal, CorrectionProposal: proposal,
@@ -85,11 +85,11 @@ func (m deterministicFakeModule) ReplacementPolicy() proposals.ReplacementPolicy
func (m deterministicFakeModule) Validators() []Validator { return nil } func (m deterministicFakeModule) Validators() []Validator { return nil }
func (m deterministicFakeModule) Propose(ctx context.Context, req ProposalRequest) ([]proposals.CorrectionProposal, error) { func (m deterministicFakeModule) Propose(ctx context.Context, req ProposalRequest) (ProposalResult, error) {
_ = ctx _ = ctx
if req.WorkingTranscript == nil || req.Section == nil { if req.WorkingTranscript == nil || req.Section == nil {
return []proposals.CorrectionProposal{}, nil return ProposalResult{}, nil
} }
out := make([]proposals.CorrectionProposal, 0) out := make([]proposals.CorrectionProposal, 0)
@@ -110,7 +110,7 @@ func (m deterministicFakeModule) Propose(ctx context.Context, req ProposalReques
} }
} }
return out, nil return ProposalResult{Proposals: out}, nil
} }
func TestChunkProposalMetadataAssociation(t *testing.T) { func TestChunkProposalMetadataAssociation(t *testing.T) {

View File

@@ -11,6 +11,7 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals" "gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/responseschema" "gitea.maximumdirect.net/eric/audita/internal/framework/responseschema"
"gitea.maximumdirect.net/eric/audita/internal/framework/validators" "gitea.maximumdirect.net/eric/audita/internal/framework/validators"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
) )
// StructuredLLMClient provides provider-agnostic structured completion. // StructuredLLMClient provides provider-agnostic structured completion.
@@ -28,7 +29,7 @@ type TranscriptModule interface {
Key() string Key() string
ReplacementPolicy() proposals.ReplacementPolicy ReplacementPolicy() proposals.ReplacementPolicy
Validators() []Validator Validators() []Validator
Propose(ctx context.Context, req ProposalRequest) ([]proposals.CorrectionProposal, error) Propose(ctx context.Context, req ProposalRequest) (ProposalResult, error)
} }
// Validator evaluates candidate proposals and returns one decision per proposal index. // Validator evaluates candidate proposals and returns one decision per proposal index.
@@ -103,6 +104,11 @@ type ProposalRequest struct {
LLMScheduler LLMScheduler `json:"-"` LLMScheduler LLMScheduler `json:"-"`
} }
type ProposalResult struct {
Proposals []proposals.CorrectionProposal `json:"proposals,omitempty"`
Warnings []stagewarnings.StageWarning `json:"warnings,omitempty"`
}
// ValidationRequest is the input to validator execution. // ValidationRequest is the input to validator execution.
type ValidationRequest = validators.Request type ValidationRequest = validators.Request

View File

@@ -50,11 +50,13 @@ func (f *fakeModule) Validators() []Validator {
return []Validator{&fakeValidator{}} return []Validator{&fakeValidator{}}
} }
func (f *fakeModule) Propose(ctx context.Context, req ProposalRequest) ([]proposals.CorrectionProposal, error) { func (f *fakeModule) Propose(ctx context.Context, req ProposalRequest) (ProposalResult, error) {
_ = ctx _ = ctx
_ = req _ = req
return []proposals.CorrectionProposal{ return ProposalResult{
{TargetSegmentID: 1, OriginalText: "a", CorrectedText: "b", Confidence: 0.9}, Proposals: []proposals.CorrectionProposal{
{TargetSegmentID: 1, OriginalText: "a", CorrectedText: "b", Confidence: 0.9},
},
}, nil }, nil
} }
@@ -68,12 +70,12 @@ func TestInterfaceContractsCompileWithFakes(t *testing.T) {
t.Fatalf("unexpected module key: %q", got) t.Fatalf("unexpected module key: %q", got)
} }
proposalsOut, err := module.Propose(context.Background(), ProposalRequest{}) proposalResult, err := module.Propose(context.Background(), ProposalRequest{})
if err != nil { if err != nil {
t.Fatalf("unexpected propose error: %v", err) t.Fatalf("unexpected propose error: %v", err)
} }
if len(proposalsOut) != 1 { if len(proposalResult.Proposals) != 1 {
t.Fatalf("expected one proposal, got %d", len(proposalsOut)) t.Fatalf("expected one proposal, got %d", len(proposalResult.Proposals))
} }
} }

View File

@@ -0,0 +1,17 @@
package llm
import "gitea.maximumdirect.net/eric/audita/internal/core/config"
// ConfiguredSecrets returns all configured LLM API-key values that should be
// redacted from diagnostics and surfaced error payloads.
func ConfiguredSecrets(cfg *config.Config) []string {
if cfg == nil {
return nil
}
effectiveValidation := cfg.EffectiveValidationLLMConfig()
return []string{
cfg.PrimaryLLM.APIKey,
cfg.ValidationLLM.APIKey,
effectiveValidation.APIKey,
}
}

View File

@@ -0,0 +1,38 @@
package llm
import (
"reflect"
"testing"
"gitea.maximumdirect.net/eric/audita/internal/core/config"
)
func TestConfiguredSecretsNilConfig(t *testing.T) {
if got := ConfiguredSecrets(nil); got != nil {
t.Fatalf("expected nil secrets for nil config, got %v", got)
}
}
func TestConfiguredSecretsWithValidationOverride(t *testing.T) {
cfg := config.Default()
cfg.PrimaryLLM.APIKey = "primary-secret"
cfg.ValidationLLM.APIKey = "validation-secret"
got := ConfiguredSecrets(&cfg)
want := []string{"primary-secret", "validation-secret", "validation-secret"}
if !reflect.DeepEqual(got, want) {
t.Fatalf("unexpected secrets: got=%v want=%v", got, want)
}
}
func TestConfiguredSecretsWithInheritedValidationKey(t *testing.T) {
cfg := config.Default()
cfg.PrimaryLLM.APIKey = "primary-secret"
cfg.ValidationLLM.APIKey = ""
got := ConfiguredSecrets(&cfg)
want := []string{"primary-secret", "", "primary-secret"}
if !reflect.DeepEqual(got, want) {
t.Fatalf("unexpected inherited secrets: got=%v want=%v", got, want)
}
}

View File

@@ -6,6 +6,7 @@ import (
"strings" "strings"
"gitea.maximumdirect.net/eric/audita/internal/core/config" "gitea.maximumdirect.net/eric/audita/internal/core/config"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
"gitea.maximumdirect.net/eric/audita/internal/core/schema" "gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts" "gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
glossarymodule "gitea.maximumdirect.net/eric/audita/internal/modules/glossary" glossarymodule "gitea.maximumdirect.net/eric/audita/internal/modules/glossary"
@@ -15,28 +16,20 @@ import (
) )
const ( const (
ModuleKeyGlossary = "glossary" ModuleKeyGlossary = modulecatalog.KeyGlossary
ModuleKeyHomophones = "homophones" ModuleKeyHomophones = modulecatalog.KeyHomophones
ModuleKeySpokenWord = "spoken_word" ModuleKeySpokenWord = modulecatalog.KeySpokenWord
ModuleKeyGrammar = "grammar" ModuleKeyGrammar = modulecatalog.KeyGrammar
) )
const ( const (
ReasonUnsupportedModule = "unsupported_module" ReasonUnsupportedModule = "unsupported_module"
) )
var knownModuleKeys = map[string]struct{}{
ModuleKeyGlossary: {},
ModuleKeyHomophones: {},
ModuleKeySpokenWord: {},
ModuleKeyGrammar: {},
}
// IsKnownModuleKey reports whether a module key is recognized by the production // IsKnownModuleKey reports whether a module key is recognized by the production
// registry scaffold. // registry scaffold.
func IsKnownModuleKey(key string) bool { func IsKnownModuleKey(key string) bool {
_, ok := knownModuleKeys[strings.TrimSpace(key)] return modulecatalog.IsSupported(key)
return ok
} }
// Dependencies holds explicit constructor dependencies for module creation. // Dependencies holds explicit constructor dependencies for module creation.
@@ -69,7 +62,7 @@ type Factory struct {
func NewFactory(deps Dependencies) *Factory { func NewFactory(deps Dependencies) *Factory {
factory := &Factory{ factory := &Factory{
deps: deps, deps: deps,
constructors: make(map[string]Constructor, len(knownModuleKeys)), constructors: make(map[string]Constructor, len(modulecatalog.SupportedKeys())),
} }
_ = factory.RegisterConstructor(ModuleKeyGlossary, constructGlossaryModule) _ = factory.RegisterConstructor(ModuleKeyGlossary, constructGlossaryModule)
_ = factory.RegisterConstructor(ModuleKeyHomophones, constructHomophonesModule) _ = factory.RegisterConstructor(ModuleKeyHomophones, constructHomophonesModule)

View File

@@ -6,6 +6,7 @@ import (
"strings" "strings"
"testing" "testing"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts" "gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals" "gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
) )
@@ -19,14 +20,14 @@ func (m noopModule) ReplacementPolicy() proposals.ReplacementPolicy {
return proposals.ReplacementPolicyRequireUnique return proposals.ReplacementPolicyRequireUnique
} }
func (m noopModule) Validators() []contracts.Validator { return nil } func (m noopModule) Validators() []contracts.Validator { return nil }
func (m noopModule) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { func (m noopModule) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
_ = ctx _ = ctx
_ = req _ = req
return nil, nil return contracts.ProposalResult{}, nil
} }
func TestKnownModuleKeyRecognition(t *testing.T) { func TestKnownModuleKeyRecognition(t *testing.T) {
for _, key := range []string{ModuleKeyGlossary, ModuleKeyHomophones, ModuleKeySpokenWord, ModuleKeyGrammar} { for _, key := range modulecatalog.SupportedKeys() {
if !IsKnownModuleKey(key) { if !IsKnownModuleKey(key) {
t.Fatalf("expected key %q to be recognized", key) t.Fatalf("expected key %q to be recognized", key)
} }

View File

@@ -1,10 +1,11 @@
package cli package processreport
import ( import (
"path/filepath" "path/filepath"
"sort" "sort"
"gitea.maximumdirect.net/eric/audita/internal/framework/runner" "gitea.maximumdirect.net/eric/audita/internal/framework/runner"
validatormetadata "gitea.maximumdirect.net/eric/audita/internal/validators/metadata"
) )
const ( const (
@@ -14,7 +15,12 @@ const (
correctionDispositionFailed = "failed" correctionDispositionFailed = "failed"
) )
type correctionLedgerEntry struct { type CorrectionLedgerInput struct {
RunDirectoryPath string
RunOutput *runner.RunOutput
}
type CorrectionLedgerEntry struct {
RunID string `json:"run_id,omitempty"` RunID string `json:"run_id,omitempty"`
ModuleKey string `json:"module_key"` ModuleKey string `json:"module_key"`
ModuleInstance string `json:"module_instance"` ModuleInstance string `json:"module_instance"`
@@ -27,33 +33,28 @@ type correctionLedgerEntry struct {
Disposition string `json:"disposition"` Disposition string `json:"disposition"`
DispositionReasonCode string `json:"disposition_reason_code,omitempty"` DispositionReasonCode string `json:"disposition_reason_code,omitempty"`
DispositionMessage string `json:"disposition_message,omitempty"` DispositionMessage string `json:"disposition_message,omitempty"`
DeterministicValidatorResults []ledgerValidatorDecisionRecord `json:"deterministic_validator_decisions,omitempty"` DeterministicValidatorResults []LedgerValidatorDecisionRecord `json:"deterministic_validator_decisions,omitempty"`
LLMValidatorResults []ledgerValidatorDecisionRecord `json:"llm_validator_decisions,omitempty"` LLMValidatorResults []LedgerValidatorDecisionRecord `json:"llm_validator_decisions,omitempty"`
} }
type ledgerValidatorDecisionRecord struct { type LedgerValidatorDecisionRecord struct {
ValidatorKey string `json:"validator_key"` ValidatorKey string `json:"validator_key"`
Approved bool `json:"approved"` Approved bool `json:"approved"`
ReasonCode string `json:"reason_code"` ReasonCode string `json:"reason_code"`
Message string `json:"message,omitempty"` Message string `json:"message,omitempty"`
} }
func buildCorrectionLedger(runDirPath string, runOutput *runner.RunOutput) []correctionLedgerEntry { func BuildCorrectionLedger(input CorrectionLedgerInput) []CorrectionLedgerEntry {
runOutput := input.RunOutput
if runOutput == nil || len(runOutput.ModuleResults) == 0 { if runOutput == nil || len(runOutput.ModuleResults) == 0 {
return nil return nil
} }
runID := "" runID := ""
if runDirPath != "" { if input.RunDirectoryPath != "" {
runID = filepath.Base(runDirPath) runID = filepath.Base(input.RunDirectoryPath)
}
entries := make([]correctionLedgerEntry, 0)
llmBacked := map[string]bool{
"spoken_form_plausibility": true,
"meaning_reversal_review": true,
"editorial_review": true,
} }
entries := make([]CorrectionLedgerEntry, 0)
for _, module := range runOutput.ModuleResults { for _, module := range runOutput.ModuleResults {
decisionsByProposal := make(map[int][]runner.ValidatorDecisionRecord) decisionsByProposal := make(map[int][]runner.ValidatorDecisionRecord)
for _, decision := range module.ValidatorDecisions { for _, decision := range module.ValidatorDecisions {
@@ -61,7 +62,7 @@ func buildCorrectionLedger(runDirPath string, runOutput *runner.RunOutput) []cor
} }
for _, change := range module.AppliedChanges { for _, change := range module.AppliedChanges {
entries = append(entries, correctionLedgerEntry{ entries = append(entries, CorrectionLedgerEntry{
RunID: runID, RunID: runID,
ModuleKey: module.ModuleKey, ModuleKey: module.ModuleKey,
ModuleInstance: module.ModuleInstance, ModuleInstance: module.ModuleInstance,
@@ -72,12 +73,12 @@ func buildCorrectionLedger(runDirPath string, runOutput *runner.RunOutput) []cor
AppliedCorrectedText: change.CorrectedText, AppliedCorrectedText: change.CorrectedText,
ReplacementPolicy: string(module.ReplacementPolicy), ReplacementPolicy: string(module.ReplacementPolicy),
Disposition: correctionDispositionApplied, Disposition: correctionDispositionApplied,
DeterministicValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], false, llmBacked), DeterministicValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], false),
LLMValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], true, llmBacked), LLMValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], true),
}) })
} }
for _, change := range module.SkippedChanges { for _, change := range module.SkippedChanges {
entries = append(entries, correctionLedgerEntry{ entries = append(entries, CorrectionLedgerEntry{
RunID: runID, RunID: runID,
ModuleKey: module.ModuleKey, ModuleKey: module.ModuleKey,
ModuleInstance: module.ModuleInstance, ModuleInstance: module.ModuleInstance,
@@ -89,12 +90,12 @@ func buildCorrectionLedger(runDirPath string, runOutput *runner.RunOutput) []cor
Disposition: correctionDispositionSkipped, Disposition: correctionDispositionSkipped,
DispositionReasonCode: string(change.SkipReason), DispositionReasonCode: string(change.SkipReason),
DispositionMessage: change.Message, DispositionMessage: change.Message,
DeterministicValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], false, llmBacked), DeterministicValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], false),
LLMValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], true, llmBacked), LLMValidatorResults: filterLedgerDecisions(decisionsByProposal[change.ProposalIndex], true),
}) })
} }
for _, rejection := range module.ValidatorRejected { for _, rejection := range module.ValidatorRejected {
entries = append(entries, correctionLedgerEntry{ entries = append(entries, CorrectionLedgerEntry{
RunID: runID, RunID: runID,
ModuleKey: module.ModuleKey, ModuleKey: module.ModuleKey,
ModuleInstance: module.ModuleInstance, ModuleInstance: module.ModuleInstance,
@@ -106,12 +107,12 @@ func buildCorrectionLedger(runDirPath string, runOutput *runner.RunOutput) []cor
Disposition: correctionDispositionRejected, Disposition: correctionDispositionRejected,
DispositionReasonCode: rejection.ReasonCode, DispositionReasonCode: rejection.ReasonCode,
DispositionMessage: rejection.Message, DispositionMessage: rejection.Message,
DeterministicValidatorResults: filterLedgerDecisions(decisionsByProposal[rejection.ProposalIndex], false, llmBacked), DeterministicValidatorResults: filterLedgerDecisions(decisionsByProposal[rejection.ProposalIndex], false),
LLMValidatorResults: filterLedgerDecisions(decisionsByProposal[rejection.ProposalIndex], true, llmBacked), LLMValidatorResults: filterLedgerDecisions(decisionsByProposal[rejection.ProposalIndex], true),
}) })
} }
if module.Status == runner.ModuleStatusFailed { if module.Status == runner.ModuleStatusFailed {
entries = append(entries, correctionLedgerEntry{ entries = append(entries, CorrectionLedgerEntry{
RunID: runID, RunID: runID,
ModuleKey: module.ModuleKey, ModuleKey: module.ModuleKey,
ModuleInstance: module.ModuleInstance, ModuleInstance: module.ModuleInstance,
@@ -135,16 +136,29 @@ func buildCorrectionLedger(runDirPath string, runOutput *runner.RunOutput) []cor
return entries return entries
} }
func filterLedgerDecisions(in []runner.ValidatorDecisionRecord, wantLLM bool, llmBacked map[string]bool) []ledgerValidatorDecisionRecord { func HasSkippedCorrections(runOutput *runner.RunOutput) bool {
if runOutput == nil {
return false
}
for _, mr := range runOutput.ModuleResults {
if len(mr.SkippedChanges) > 0 || len(mr.ValidatorRejected) > 0 {
return true
}
}
return false
}
func filterLedgerDecisions(in []runner.ValidatorDecisionRecord, wantLLM bool) []LedgerValidatorDecisionRecord {
if len(in) == 0 { if len(in) == 0 {
return nil return nil
} }
out := make([]ledgerValidatorDecisionRecord, 0, len(in)) out := make([]LedgerValidatorDecisionRecord, 0, len(in))
for _, decision := range in { for _, decision := range in {
if llmBacked[decision.ValidatorName] != wantLLM { isLLMBacked := validatormetadata.ClassForKey(decision.ValidatorName) == validatormetadata.ExecutionClassLLMBacked
if isLLMBacked != wantLLM {
continue continue
} }
out = append(out, ledgerValidatorDecisionRecord{ out = append(out, LedgerValidatorDecisionRecord{
ValidatorKey: decision.ValidatorName, ValidatorKey: decision.ValidatorName,
Approved: decision.Approved, Approved: decision.Approved,
ReasonCode: decision.ReasonCode, ReasonCode: decision.ReasonCode,

View File

@@ -0,0 +1,128 @@
package processreport
import (
"testing"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/runner"
)
func TestBuildCorrectionLedgerClassifiesValidatorDecisionsFromCanonicalMetadata(t *testing.T) {
output := &runner.RunOutput{
ModuleResults: []runner.ModuleResult{
{
ModuleKey: "glossary",
ModuleInstance: "glossary",
ReplacementPolicy: proposals.ReplacementPolicyReplaceAll,
ValidatorDecisions: []runner.ValidatorDecisionRecord{
{ValidatorName: "proposal_shape", ProposalIndex: 3, Approved: true, ReasonCode: "approved"},
{ValidatorName: "spoken_form_plausibility", ProposalIndex: 3, Approved: true, ReasonCode: "approved"},
},
AppliedChanges: []proposals.AppliedChange{
{
ProposalIndex: 3,
ModuleKey: "glossary",
ModuleInstance: "glossary",
TargetSegmentID: 1,
OriginalText: "gestures",
CorrectedText: "Jesters",
},
},
},
},
}
ledger := BuildCorrectionLedger(CorrectionLedgerInput{
RunDirectoryPath: "/tmp/audita-run-id",
RunOutput: output,
})
if len(ledger) != 1 {
t.Fatalf("expected one ledger entry, got %d", len(ledger))
}
entry := ledger[0]
if entry.RunID != "audita-run-id" || entry.Disposition != "applied" || entry.AppliedCorrectedText != "Jesters" {
t.Fatalf("unexpected applied ledger entry: %+v", entry)
}
if len(entry.DeterministicValidatorResults) != 1 || entry.DeterministicValidatorResults[0].ValidatorKey != "proposal_shape" {
t.Fatalf("unexpected deterministic decision split: %+v", entry.DeterministicValidatorResults)
}
if len(entry.LLMValidatorResults) != 1 || entry.LLMValidatorResults[0].ValidatorKey != "spoken_form_plausibility" {
t.Fatalf("unexpected llm-backed decision split: %+v", entry.LLMValidatorResults)
}
}
func TestBuildCorrectionLedgerPreservesDispositionPolicy(t *testing.T) {
output := &runner.RunOutput{
ModuleResults: []runner.ModuleResult{
{
ModuleKey: "grammar",
ModuleInstance: "grammar",
ReplacementPolicy: proposals.ReplacementPolicyRequireUnique,
Status: runner.ModuleStatusSuccess,
SkippedChanges: []proposals.SkippedChange{
{
ProposalIndex: 2,
TargetSegmentID: 7,
OriginalText: "old",
CorrectedText: "new",
SkipReason: proposals.SkipReasonAmbiguousOriginal,
Message: "ambiguous",
},
},
ValidatorRejected: []runner.ValidatorRejectedChange{
{
ProposalIndex: 3,
TargetSegmentID: 8,
OriginalText: "before",
CorrectedText: "after",
ReasonCode: "protected_term",
Message: "blocked",
},
},
},
{
ModuleKey: "capitalization",
ModuleInstance: "capitalization",
Status: runner.ModuleStatusFailed,
ErrorMessage: "failed",
},
},
}
ledger := BuildCorrectionLedger(CorrectionLedgerInput{RunOutput: output})
if len(ledger) != 3 {
t.Fatalf("expected skipped, rejected, and failed entries, got %+v", ledger)
}
byDisposition := make(map[string]CorrectionLedgerEntry)
for _, entry := range ledger {
byDisposition[entry.Disposition] = entry
}
if byDisposition["skipped"].DispositionReasonCode != string(proposals.SkipReasonAmbiguousOriginal) ||
byDisposition["skipped"].DispositionMessage != "ambiguous" {
t.Fatalf("unexpected skipped ledger entry: %+v", byDisposition["skipped"])
}
if byDisposition["rejected"].DispositionReasonCode != "protected_term" ||
byDisposition["rejected"].ProposedCorrectedText != "after" {
t.Fatalf("unexpected rejected ledger entry: %+v", byDisposition["rejected"])
}
if byDisposition["failed"].DispositionReasonCode != "module_failed" ||
byDisposition["failed"].DispositionMessage != "failed" {
t.Fatalf("unexpected failed ledger entry: %+v", byDisposition["failed"])
}
}
func TestHasSkippedCorrectionsIncludesApplicationSkipsAndValidatorRejections(t *testing.T) {
if HasSkippedCorrections(nil) {
t.Fatal("nil output should not have skipped corrections")
}
if HasSkippedCorrections(&runner.RunOutput{ModuleResults: []runner.ModuleResult{{AppliedChanges: []proposals.AppliedChange{{ProposalIndex: 1}}}}}) {
t.Fatal("applied-only output should not have skipped corrections")
}
if !HasSkippedCorrections(&runner.RunOutput{ModuleResults: []runner.ModuleResult{{SkippedChanges: []proposals.SkippedChange{{ProposalIndex: 1}}}}}) {
t.Fatal("application skips should count as skipped corrections")
}
if !HasSkippedCorrections(&runner.RunOutput{ModuleResults: []runner.ModuleResult{{ValidatorRejected: []runner.ValidatorRejectedChange{{ProposalIndex: 1}}}}}) {
t.Fatal("validator rejections should count as skipped corrections")
}
}

View File

@@ -0,0 +1,158 @@
package processreport
import (
"time"
"gitea.maximumdirect.net/eric/audita/internal/core/chunking"
"gitea.maximumdirect.net/eric/audita/internal/core/diagnostics"
"gitea.maximumdirect.net/eric/audita/internal/core/normalization"
"gitea.maximumdirect.net/eric/audita/internal/core/reporting"
"gitea.maximumdirect.net/eric/audita/internal/framework/runner"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
)
// BuildInput contains already-computed process execution facts for report assembly.
type BuildInput struct {
Status string
TranscriptPath string
GlossaryPath string
OutputPath string
Modules []string
OutputSchema string
ConfigVersion *int
StartedAt time.Time
CompletedAt time.Time
ErrorMessage string
ErrorPhase string
RunDirectoryPath string
NormalizationSummary *normalization.NormalizationSummary
ChunkingSummary *chunking.Summary
RunOutput *runner.RunOutput
}
// Build creates the public process report without owning command parsing or config loading.
func Build(input BuildInput) reporting.ProcessReport {
report := reporting.ProcessReport{
ReportMetadata: reporting.ReportMetadata{
ReportSchemaName: reporting.DefaultProcessReportSchemaName,
ReportSchemaVersion: reporting.DefaultProcessReportSchemaVersion,
OutputSchema: input.OutputSchema,
ConfigVersion: input.ConfigVersion,
},
Phase: "default_pipeline",
Status: input.Status,
Operation: "process",
TranscriptPath: input.TranscriptPath,
GlossaryPath: input.GlossaryPath,
OutputPath: input.OutputPath,
Modules: append([]string(nil), input.Modules...),
StartedAt: input.StartedAt,
CompletedAt: &input.CompletedAt,
ErrorPhase: input.ErrorPhase,
}
if input.RunDirectoryPath != "" {
runSucceeded := input.Status == "success"
metadata := diagnostics.BuildDiagnosticsMetadata(input.RunDirectoryPath, runSucceeded)
report.Diagnostics = &metadata
}
if input.ErrorMessage != "" {
report.ErrorMessage = input.ErrorMessage
}
if input.NormalizationSummary != nil {
report.InputSegmentCount = &input.NormalizationSummary.InputSegmentCount
report.NormalizedSegmentCount = &input.NormalizationSummary.OutputSegmentCount
report.NormalizationMerges = &input.NormalizationSummary.MergesPerformed
report.NormalizationIDReassignments = &input.NormalizationSummary.IDsReassigned
report.NormalizationSkipped.DifferentSpeakers = &input.NormalizationSummary.SkippedMerges.DifferentSpeakers
report.NormalizationSkipped.GapTooLarge = &input.NormalizationSummary.SkippedMerges.GapTooLarge
report.NormalizationSkipped.DurationExceeded = &input.NormalizationSummary.SkippedMerges.DurationExceeded
report.NormalizationSkipped.TokenLimitExceeded = &input.NormalizationSummary.SkippedMerges.TokenLimitExceeded
}
if input.ChunkingSummary != nil {
report.Chunking = &reporting.ChunkingSummary{
ChunkCount: input.ChunkingSummary.ChunkCount,
MinEstimatedTokens: input.ChunkingSummary.MinEstimatedTokens,
MaxEstimatedTokens: input.ChunkingSummary.MaxEstimatedTokens,
TotalEstimatedTokens: input.ChunkingSummary.TotalEstimatedTokens,
TargetSections: input.ChunkingSummary.TargetSections,
MaxSectionTokens: input.ChunkingSummary.MaxSectionTokens,
MinSectionTokens: input.ChunkingSummary.MinSectionTokens,
}
}
report.ModulesSummary, report.ModuleResults = buildModuleReporting(input.RunOutput)
return report
}
func buildModuleReporting(runOutput *runner.RunOutput) (*reporting.ModulesSummary, []reporting.ModuleReport) {
if runOutput == nil || len(runOutput.ModuleResults) == 0 {
return nil, nil
}
moduleReports := make([]reporting.ModuleReport, 0, len(runOutput.ModuleResults))
summary := &reporting.ModulesSummary{ModuleCount: len(runOutput.ModuleResults)}
for _, r := range runOutput.ModuleResults {
startedAt := r.StartedAt
completedAt := r.CompletedAt
moduleReports = append(moduleReports, reporting.ModuleReport{
ModuleKey: r.ModuleKey,
ModuleInstance: r.ModuleInstance,
ReplacementPolicy: string(r.ReplacementPolicy),
Status: r.Status,
ProposalCount: r.ProposalCount,
Warnings: append([]stagewarnings.StageWarning(nil), r.Warnings...),
ValidatorDecisions: mapValidatorDecisions(r.ValidatorDecisions),
ValidatorRejected: mapValidatorRejected(r.ValidatorRejected),
AppliedChanges: r.AppliedChanges,
SkippedChanges: r.SkippedChanges,
ErrorMessage: r.ErrorMessage,
StartedAt: &startedAt,
CompletedAt: &completedAt,
})
summary.TotalAppliedChanges += len(r.AppliedChanges)
summary.TotalSkippedChanges += len(r.SkippedChanges) + len(r.ValidatorRejected)
if r.Status == runner.ModuleStatusFailed && summary.FailedModuleInstance == "" {
summary.FailedModuleInstance = r.ModuleInstance
}
}
return summary, moduleReports
}
func mapValidatorDecisions(in []runner.ValidatorDecisionRecord) []reporting.ValidatorDecisionReport {
if len(in) == 0 {
return nil
}
out := make([]reporting.ValidatorDecisionReport, len(in))
for i, d := range in {
out[i] = reporting.ValidatorDecisionReport{
ValidatorName: d.ValidatorName,
ProposalIndex: d.ProposalIndex,
Approved: d.Approved,
ReasonCode: d.ReasonCode,
Message: d.Message,
DiagnosticArtifactPath: d.DiagnosticArtifactPath,
}
}
return out
}
func mapValidatorRejected(in []runner.ValidatorRejectedChange) []reporting.ValidatorRejectedReport {
if len(in) == 0 {
return nil
}
out := make([]reporting.ValidatorRejectedReport, len(in))
for i, d := range in {
out[i] = reporting.ValidatorRejectedReport{
ValidatorName: d.ValidatorName,
ProposalIndex: d.ProposalIndex,
ModuleKey: d.ModuleKey,
ModuleInstance: d.ModuleInstance,
TargetSegmentID: d.TargetSegmentID,
OriginalText: d.OriginalText,
CorrectedText: d.CorrectedText,
ReasonCode: d.ReasonCode,
Message: d.Message,
}
}
return out
}

View File

@@ -0,0 +1,180 @@
package processreport
import (
"path/filepath"
"testing"
"time"
"gitea.maximumdirect.net/eric/audita/internal/core/chunking"
"gitea.maximumdirect.net/eric/audita/internal/core/diagnostics"
"gitea.maximumdirect.net/eric/audita/internal/core/normalization"
"gitea.maximumdirect.net/eric/audita/internal/core/reporting"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/runner"
)
func TestBuildSuccessReportMapsExecutionFacts(t *testing.T) {
startedAt := time.Date(2026, 5, 23, 10, 0, 0, 0, time.UTC)
completedAt := startedAt.Add(time.Second)
configVersion := 4
targetSections := 2
inputSegments := 5
outputSegments := 4
merges := 1
reassigned := 2
differentSpeakers := 3
report := Build(BuildInput{
Status: "success",
TranscriptPath: "transcript.json",
GlossaryPath: "glossary.yaml",
OutputPath: "out.json",
Modules: []string{"grammar"},
OutputSchema: "default",
ConfigVersion: &configVersion,
StartedAt: startedAt,
CompletedAt: completedAt,
RunDirectoryPath: filepath.Join(
"tmp",
"audita-run",
),
NormalizationSummary: &normalization.NormalizationSummary{
InputSegmentCount: inputSegments,
OutputSegmentCount: outputSegments,
MergesPerformed: merges,
IDsReassigned: reassigned,
SkippedMerges: struct {
DifferentSpeakers int `json:"different_speakers"`
GapTooLarge int `json:"gap_too_large"`
DurationExceeded int `json:"duration_exceeded"`
TokenLimitExceeded int `json:"token_limit_exceeded"`
}{
DifferentSpeakers: differentSpeakers,
},
},
ChunkingSummary: &chunking.Summary{
ChunkCount: 3,
MinEstimatedTokens: 10,
MaxEstimatedTokens: 20,
TotalEstimatedTokens: 45,
TargetSections: &targetSections,
MaxSectionTokens: 200,
MinSectionTokens: 50,
},
RunOutput: &runner.RunOutput{
ModuleResults: []runner.ModuleResult{
{
ModuleKey: "grammar",
ModuleInstance: "grammar",
ReplacementPolicy: proposals.ReplacementPolicyRequireUnique,
Status: runner.ModuleStatusSuccess,
ProposalCount: 2,
ValidatorDecisions: []runner.ValidatorDecisionRecord{
{
ValidatorName: "proposal_shape",
ProposalIndex: 1,
Approved: true,
ReasonCode: "approved",
Message: "ok",
DiagnosticArtifactPath: "diagnostics/validator.json",
},
},
ValidatorRejected: []runner.ValidatorRejectedChange{
{
ValidatorName: "protected_term",
ProposalIndex: 2,
ModuleKey: "grammar",
ModuleInstance: "grammar",
TargetSegmentID: 7,
OriginalText: "old",
CorrectedText: "new",
ReasonCode: "protected_term",
Message: "blocked",
},
},
AppliedChanges: []proposals.AppliedChange{
{ProposalIndex: 1, TargetSegmentID: 7, OriginalText: "old", CorrectedText: "new"},
},
SkippedChanges: []proposals.SkippedChange{
{ProposalIndex: 3, TargetSegmentID: 8, SkipReason: proposals.SkipReasonMissingSegment},
},
StartedAt: startedAt,
CompletedAt: completedAt,
},
},
},
})
if report.ReportMetadata.ReportSchemaName != reporting.DefaultProcessReportSchemaName ||
report.ReportMetadata.ReportSchemaVersion != reporting.DefaultProcessReportSchemaVersion ||
report.ReportMetadata.OutputSchema != "default" ||
report.ReportMetadata.ConfigVersion == nil ||
*report.ReportMetadata.ConfigVersion != configVersion {
t.Fatalf("unexpected report metadata: %+v", report.ReportMetadata)
}
if report.Phase != "default_pipeline" || report.Operation != "process" || report.Status != "success" {
t.Fatalf("unexpected process identity fields: phase=%q operation=%q status=%q", report.Phase, report.Operation, report.Status)
}
if report.Diagnostics == nil || report.Diagnostics.CorrectionLedgerPath != filepath.Join("tmp", "audita-run", diagnostics.ArtifactCorrectionLedger) {
t.Fatalf("unexpected diagnostics metadata: %+v", report.Diagnostics)
}
if report.InputSegmentCount == nil || *report.InputSegmentCount != inputSegments ||
report.NormalizationSkipped.DifferentSpeakers == nil ||
*report.NormalizationSkipped.DifferentSpeakers != differentSpeakers {
t.Fatalf("unexpected normalization summary: %+v", report)
}
if report.Chunking == nil || report.Chunking.ChunkCount != 3 || report.Chunking.TargetSections == nil || *report.Chunking.TargetSections != targetSections {
t.Fatalf("unexpected chunking summary: %+v", report.Chunking)
}
if report.ModulesSummary == nil ||
report.ModulesSummary.ModuleCount != 1 ||
report.ModulesSummary.TotalAppliedChanges != 1 ||
report.ModulesSummary.TotalSkippedChanges != 2 {
t.Fatalf("unexpected modules summary: %+v", report.ModulesSummary)
}
if len(report.ModuleResults) != 1 ||
len(report.ModuleResults[0].ValidatorDecisions) != 1 ||
len(report.ModuleResults[0].ValidatorRejected) != 1 {
t.Fatalf("unexpected module reports: %+v", report.ModuleResults)
}
}
func TestBuildFailedReportPreservesErrorAndFailureSummary(t *testing.T) {
startedAt := time.Date(2026, 5, 23, 10, 0, 0, 0, time.UTC)
completedAt := startedAt.Add(time.Second)
report := Build(BuildInput{
Status: "failed",
TranscriptPath: "transcript.json",
GlossaryPath: "glossary.yaml",
Modules: []string{"grammar"},
OutputSchema: "default",
StartedAt: startedAt,
CompletedAt: completedAt,
ErrorPhase: "module",
ErrorMessage: "module failed",
RunDirectoryPath: "run-dir",
RunOutput: &runner.RunOutput{
ModuleResults: []runner.ModuleResult{
{
ModuleKey: "grammar",
ModuleInstance: "grammar",
Status: runner.ModuleStatusFailed,
ErrorMessage: "module failed",
StartedAt: startedAt,
CompletedAt: completedAt,
},
},
},
})
if report.Status != "failed" || report.ErrorPhase != "module" || report.ErrorMessage != "module failed" {
t.Fatalf("unexpected failure fields: %+v", report)
}
if report.Diagnostics == nil || report.Diagnostics.ErrorLogPath == "" {
t.Fatalf("expected failure diagnostics metadata, got %+v", report.Diagnostics)
}
if report.ModulesSummary == nil || report.ModulesSummary.FailedModuleInstance != "grammar" {
t.Fatalf("unexpected failed module summary: %+v", report.ModulesSummary)
}
}

View File

@@ -0,0 +1,45 @@
package promptcontext
import (
"encoding/json"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
)
type transcriptSectionSegment struct {
ID int `json:"id"`
Speaker string `json:"speaker"`
Start float64 `json:"start"`
End float64 `json:"end"`
Text string `json:"text"`
Categories []string `json:"categories,omitempty"`
}
type transcriptSectionPayload struct {
SectionIndex int `json:"section_index"`
Segments []transcriptSectionSegment `json:"segments"`
}
// MarshalTranscriptSectionJSON builds the standardized transcript-section JSON
// payload consumed by proposal prompt templates.
func MarshalTranscriptSectionJSON(transcript *schema.Transcript, sectionIndex int) ([]byte, error) {
payload := transcriptSectionPayload{
SectionIndex: sectionIndex,
Segments: make([]transcriptSectionSegment, 0),
}
if transcript != nil {
for _, s := range transcript.Segments {
payload.Segments = append(payload.Segments, transcriptSectionSegment{
ID: s.ID,
Speaker: s.Speaker,
Start: s.Start,
End: s.End,
Text: s.Text,
Categories: append([]string(nil), s.Categories...),
})
}
}
return json.MarshalIndent(payload, "", " ")
}

View File

@@ -0,0 +1,95 @@
package promptcontext
import (
"encoding/json"
"reflect"
"testing"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
)
func TestMarshalTranscriptSectionJSONShape(t *testing.T) {
transcript := &schema.Transcript{Segments: []schema.Segment{
{ID: 1, Speaker: "A", Start: 0.1, End: 1.2, Text: "alpha", Categories: []string{"session", "intro"}},
}}
raw, err := MarshalTranscriptSectionJSON(transcript, 3)
if err != nil {
t.Fatalf("MarshalTranscriptSectionJSON error: %v", err)
}
var decoded map[string]any
if err := json.Unmarshal(raw, &decoded); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if got := decoded["section_index"]; got != float64(3) {
t.Fatalf("section_index: got=%v want=%v", got, 3)
}
segments, ok := decoded["segments"].([]any)
if !ok || len(segments) != 1 {
t.Fatalf("segments shape mismatch: %T %+v", decoded["segments"], decoded["segments"])
}
first, ok := segments[0].(map[string]any)
if !ok {
t.Fatalf("segment shape mismatch: %T", segments[0])
}
if first["id"] != float64(1) || first["speaker"] != "A" || first["start"] != 0.1 || first["end"] != 1.2 || first["text"] != "alpha" {
t.Fatalf("unexpected segment fields: %+v", first)
}
cats, ok := first["categories"].([]any)
if !ok || len(cats) != 2 || cats[0] != "session" || cats[1] != "intro" {
t.Fatalf("unexpected categories: %+v", first["categories"])
}
}
func TestMarshalTranscriptSectionJSONEmptyTranscript(t *testing.T) {
raw, err := MarshalTranscriptSectionJSON(nil, 0)
if err != nil {
t.Fatalf("MarshalTranscriptSectionJSON error: %v", err)
}
var decoded map[string]any
if err := json.Unmarshal(raw, &decoded); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if got := decoded["section_index"]; got != float64(0) {
t.Fatalf("section_index: got=%v want=%v", got, 0)
}
segments, ok := decoded["segments"].([]any)
if !ok {
t.Fatalf("segments shape mismatch: %T", decoded["segments"])
}
if len(segments) != 0 {
t.Fatalf("expected empty segments, got %d", len(segments))
}
}
func TestMarshalTranscriptSectionJSONCopiesCategories(t *testing.T) {
transcript := &schema.Transcript{Segments: []schema.Segment{
{ID: 1, Speaker: "A", Start: 0, End: 1, Text: "alpha", Categories: []string{"kept"}},
}}
raw, err := MarshalTranscriptSectionJSON(transcript, 1)
if err != nil {
t.Fatalf("MarshalTranscriptSectionJSON error: %v", err)
}
transcript.Segments[0].Categories[0] = "changed"
var decoded struct {
Segments []struct {
Categories []string `json:"categories"`
} `json:"segments"`
}
if err := json.Unmarshal(raw, &decoded); err != nil {
t.Fatalf("unmarshal: %v", err)
}
if len(decoded.Segments) != 1 {
t.Fatalf("expected one segment, got %d", len(decoded.Segments))
}
if !reflect.DeepEqual(decoded.Segments[0].Categories, []string{"kept"}) {
t.Fatalf("expected copied categories, got %v", decoded.Segments[0].Categories)
}
}

View File

@@ -15,6 +15,9 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/framework/llm" "gitea.maximumdirect.net/eric/audita/internal/framework/llm"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals" "gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/responseschema" "gitea.maximumdirect.net/eric/audita/internal/framework/responseschema"
"gitea.maximumdirect.net/eric/audita/internal/framework/stagename"
"gitea.maximumdirect.net/eric/audita/internal/framework/structuredoutput"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
) )
// InteractionDiagnosticsWriter writes machine-readable prompt/response artifacts. // InteractionDiagnosticsWriter writes machine-readable prompt/response artifacts.
@@ -68,6 +71,7 @@ type Request struct {
type Result struct { type Result struct {
Corrections []proposals.CorrectionProposal `json:"corrections"` Corrections []proposals.CorrectionProposal `json:"corrections"`
Enriched []proposals.EnrichedCorrectionProposal `json:"enriched"` Enriched []proposals.EnrichedCorrectionProposal `json:"enriched"`
Warnings []stagewarnings.StageWarning `json:"warnings,omitempty"`
Artifacts InteractionArtifacts `json:"artifacts,omitempty"` Artifacts InteractionArtifacts `json:"artifacts,omitempty"`
} }
@@ -92,7 +96,7 @@ func GenerateCandidates(ctx context.Context, req Request) (Result, error) {
stage := strings.TrimSpace(req.StageName) stage := strings.TrimSpace(req.StageName)
if stage == "" { if stage == "" {
stage = buildStageName(req.ModuleInstance, req.Section) stage = stagename.ProposalGeneration(req.ModuleInstance, sectionIndexPtr(req.Section))
} }
model := resolveModel(req.Config, req.Model) model := resolveModel(req.Config, req.Model)
messages := append([]contracts.LLMMessage(nil), req.Messages...) messages := append([]contracts.LLMMessage(nil), req.Messages...)
@@ -104,7 +108,7 @@ func GenerateCandidates(ctx context.Context, req Request) (Result, error) {
writer = diagnosticsWriterAdapter{ writer = diagnosticsWriterAdapter{
writer: llm.NewDiagnosticsWriter( writer: llm.NewDiagnosticsWriter(
filepath.Join(req.DiagnosticsDir, req.ModuleInstance), filepath.Join(req.DiagnosticsDir, req.ModuleInstance),
proposalGenerationSecrets(req.Config), llm.ConfiguredSecrets(req.Config),
), ),
} }
} }
@@ -141,7 +145,7 @@ func GenerateCandidates(ctx context.Context, req Request) (Result, error) {
if len(req.PromptMetadata) > 0 { if len(req.PromptMetadata) > 0 {
requestMetadata["prompt_metadata"] = req.PromptMetadata requestMetadata["prompt_metadata"] = req.PromptMetadata
} }
requestMetadata["response_schema"] = schemaMetadata(responseSchema) requestMetadata["response_schema"] = responseSchema.DiagnosticsMap()
if writer != nil { if writer != nil {
artifacts, _ = writer.WriteInteraction( artifacts, _ = writer.WriteInteraction(
@@ -156,6 +160,12 @@ func GenerateCandidates(ctx context.Context, req Request) (Result, error) {
} }
if callErr != nil { if callErr != nil {
if structuredoutput.IsMalformedError(callErr) {
return Result{
Warnings: []stagewarnings.StageWarning{newMalformedProposalWarning(req.Section, artifacts, callErr)},
Artifacts: artifacts,
}, nil
}
return Result{}, fmt.Errorf("proposal generation completion failed: %w", callErr) return Result{}, fmt.Errorf("proposal generation completion failed: %w", callErr)
} }
@@ -168,9 +178,6 @@ func GenerateCandidates(ctx context.Context, req Request) (Result, error) {
CorrectedText: raw.CorrectedText, CorrectedText: raw.CorrectedText,
Confidence: raw.Confidence, Confidence: raw.Confidence,
} }
if err := candidate.Validate(); err != nil {
return Result{}, fmt.Errorf("invalid structured correction at index %d: %w", i, err)
}
corrections = append(corrections, candidate) corrections = append(corrections, candidate)
enrichedCandidate := proposals.EnrichedCorrectionProposal{ enrichedCandidate := proposals.EnrichedCorrectionProposal{
@@ -191,27 +198,11 @@ func GenerateCandidates(ctx context.Context, req Request) (Result, error) {
return Result{ return Result{
Corrections: corrections, Corrections: corrections,
Enriched: enriched, Enriched: enriched,
Warnings: nil,
Artifacts: artifacts, Artifacts: artifacts,
}, nil }, nil
} }
func schemaMetadata(schema responseschema.Schema) map[string]any {
return map[string]any{
"id": schema.ID,
"version": schema.Version,
"name": schema.Name,
"sha256": schema.SHA256,
}
}
func buildStageName(moduleInstance string, section *contracts.SectionMetadata) string {
base := fmt.Sprintf("%s:proposal-generation", moduleInstance)
if section == nil {
return base
}
return fmt.Sprintf("%s:section-%04d", base, section.Index)
}
func resolveModel(cfg *config.Config, override string) string { func resolveModel(cfg *config.Config, override string) string {
if strings.TrimSpace(override) != "" { if strings.TrimSpace(override) != "" {
return strings.TrimSpace(override) return strings.TrimSpace(override)
@@ -222,17 +213,6 @@ func resolveModel(cfg *config.Config, override string) string {
return llm.ResolvePrimaryConfig(*cfg).Model return llm.ResolvePrimaryConfig(*cfg).Model
} }
func proposalGenerationSecrets(cfg *config.Config) []string {
if cfg == nil {
return nil
}
return []string{
cfg.PrimaryLLM.APIKey,
cfg.ValidationLLM.APIKey,
cfg.EffectiveValidationLLMConfig().APIKey,
}
}
func errPayload(err error) any { func errPayload(err error) any {
if err == nil { if err == nil {
return nil return nil
@@ -240,6 +220,35 @@ func errPayload(err error) any {
return map[string]any{"error": err.Error()} return map[string]any{"error": err.Error()}
} }
func newMalformedProposalWarning(section *contracts.SectionMetadata, artifacts InteractionArtifacts, err error) stagewarnings.StageWarning {
warning := stagewarnings.StageWarning{
Scope: stagewarnings.ScopeProposalGeneration,
ReasonCode: "proposal_response_malformed",
Message: strings.TrimSpace(err.Error()),
DiagnosticArtifactPath: diagnosticArtifactPath(artifacts),
}
if section != nil {
sectionIndex := section.Index
warning.SectionIndex = &sectionIndex
}
return warning
}
func diagnosticArtifactPath(artifacts InteractionArtifacts) string {
if artifacts.ErrorPayloadPath != "" {
return artifacts.ErrorPayloadPath
}
return artifacts.ResponsePayloadPath
}
func sectionIndexPtr(section *contracts.SectionMetadata) *int {
if section == nil {
return nil
}
index := section.Index
return &index
}
type diagnosticsWriterAdapter struct { type diagnosticsWriterAdapter struct {
writer *llm.DiagnosticsWriter writer *llm.DiagnosticsWriter
} }

View File

@@ -224,12 +224,12 @@ func TestGenerateCandidatesDiagnosticsIncludeSchemaMetadata(t *testing.T) {
} }
} }
func TestGenerateCandidatesMalformedStructuredResponse(t *testing.T) { func TestGenerateCandidatesInvalidCorrectionIsPreservedForLaterValidation(t *testing.T) {
client := &fakeStructuredClient{ client := &fakeStructuredClient{
responses: []StructuredCorrectionSet{ responses: []StructuredCorrectionSet{
{ {
Corrections: []StructuredCorrectionProposal{ Corrections: []StructuredCorrectionProposal{
{TargetSegmentID: 1, OriginalText: "x", CorrectedText: "", Confidence: 0.9}, {TargetSegmentID: 0, OriginalText: "x", CorrectedText: "", Confidence: 1.2},
}, },
}, },
}, },
@@ -237,9 +237,55 @@ func TestGenerateCandidatesMalformedStructuredResponse(t *testing.T) {
req := defaultRequest(t) req := defaultRequest(t)
req.LLMClient = client req.LLMClient = client
_, err := GenerateCandidates(context.Background(), req) result, err := GenerateCandidates(context.Background(), req)
if err == nil || !strings.Contains(err.Error(), "invalid structured correction") { if err != nil {
t.Fatalf("expected structured response validation failure, got %v", err) t.Fatalf("expected invalid correction to survive generation, got %v", err)
}
if len(result.Corrections) != 1 {
t.Fatalf("expected one correction, got %+v", result)
}
if result.Corrections[0].TargetSegmentID != 0 || result.Corrections[0].Confidence != 1.2 {
t.Fatalf("unexpected preserved correction: %+v", result.Corrections[0])
}
if len(result.Warnings) != 0 {
t.Fatalf("did not expect warnings for individually invalid corrections, got %+v", result.Warnings)
}
}
func TestGenerateCandidatesMalformedStructuredOutputReturnsWarning(t *testing.T) {
client := &fakeStructuredClient{err: errors.New("malformed structured output")}
req := defaultRequest(t)
req.LLMClient = client
result, err := GenerateCandidates(context.Background(), req)
if err != nil {
t.Fatalf("expected malformed structured output to downgrade to warning, got %v", err)
}
if len(result.Corrections) != 0 || len(result.Enriched) != 0 {
t.Fatalf("expected no proposals on malformed response, got %+v", result)
}
if len(result.Warnings) != 1 {
t.Fatalf("expected one warning, got %+v", result.Warnings)
}
if result.Warnings[0].ReasonCode != "proposal_response_malformed" {
t.Fatalf("unexpected warning: %+v", result.Warnings[0])
}
}
func TestGenerateCandidatesProviderMalformedEnvelopeReturnsWarning(t *testing.T) {
client := &fakeStructuredClient{err: errors.New("provider response missing choices")}
req := defaultRequest(t)
req.LLMClient = client
result, err := GenerateCandidates(context.Background(), req)
if err != nil {
t.Fatalf("expected malformed provider envelope to downgrade to warning, got %v", err)
}
if len(result.Corrections) != 0 || len(result.Enriched) != 0 {
t.Fatalf("expected no proposals on malformed response, got %+v", result)
}
if len(result.Warnings) != 1 || result.Warnings[0].ReasonCode != "proposal_response_malformed" {
t.Fatalf("unexpected warnings: %+v", result.Warnings)
} }
} }
@@ -315,20 +361,22 @@ func TestGenerateCandidatesMultipleSectionsStableMetadata(t *testing.T) {
} }
func TestGenerateCandidatesDiagnosticsWrittenAndRedacted(t *testing.T) { func TestGenerateCandidatesDiagnosticsWrittenAndRedacted(t *testing.T) {
secret := "proposal-secret" primarySecret := "proposal-primary-secret"
validationSecret := "proposal-validation-secret"
client := &fakeStructuredClient{ client := &fakeStructuredClient{
responses: []StructuredCorrectionSet{ responses: []StructuredCorrectionSet{
{Corrections: []StructuredCorrectionProposal{{TargetSegmentID: 1, OriginalText: secret, CorrectedText: "safe", Confidence: 0.9}}}, {Corrections: []StructuredCorrectionProposal{{TargetSegmentID: 1, OriginalText: validationSecret, CorrectedText: "safe", Confidence: 0.9}}},
}, },
} }
cfg := config.Default() cfg := config.Default()
cfg.PrimaryLLM.APIKey = secret cfg.PrimaryLLM.APIKey = primarySecret
cfg.ValidationLLM.APIKey = validationSecret
req := defaultRequest(t) req := defaultRequest(t)
req.Config = &cfg req.Config = &cfg
req.LLMClient = client req.LLMClient = client
req.DiagnosticsDir = t.TempDir() req.DiagnosticsDir = t.TempDir()
req.Messages = []contracts.LLMMessage{ req.Messages = []contracts.LLMMessage{
{Role: "system", Content: "include secret " + secret}, {Role: "system", Content: "include secret " + primarySecret},
{Role: "user", Content: "fix it"}, {Role: "user", Content: "fix it"},
} }
@@ -345,8 +393,8 @@ func TestGenerateCandidatesDiagnosticsWrittenAndRedacted(t *testing.T) {
if readErr != nil { if readErr != nil {
t.Fatalf("read artifact %q: %v", path, readErr) t.Fatalf("read artifact %q: %v", path, readErr)
} }
if strings.Contains(string(raw), secret) { if strings.Contains(string(raw), primarySecret) || strings.Contains(string(raw), validationSecret) {
t.Fatalf("artifact leaked secret %q: %s", path, string(raw)) t.Fatalf("artifact leaked configured secret in %q: %s", path, string(raw))
} }
if !strings.Contains(string(raw), "[REDACTED]") { if !strings.Contains(string(raw), "[REDACTED]") {
t.Fatalf("expected redaction marker in artifact %q: %s", path, string(raw)) t.Fatalf("expected redaction marker in artifact %q: %s", path, string(raw))

View File

@@ -0,0 +1,81 @@
package proposal_generation
import (
"context"
"fmt"
"strings"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/stagename"
"gitea.maximumdirect.net/eric/audita/internal/prompts"
)
type ProposalMessageBuilder func(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error)
type ModuleProposalRequest struct {
ProposalRequest contracts.ProposalRequest
PromptID string
BuildMessages ProposalMessageBuilder
}
// ExecuteModuleProposal runs shared proposal generation plumbing for one
// module, leaving only module-specific prompt message building at call sites.
func ExecuteModuleProposal(ctx context.Context, req ModuleProposalRequest) (contracts.ProposalResult, error) {
if req.BuildMessages == nil {
return contracts.ProposalResult{}, fmt.Errorf("proposal message builder is required")
}
if strings.TrimSpace(req.PromptID) == "" {
return contracts.ProposalResult{}, fmt.Errorf("prompt ID must not be empty")
}
sectionIndex := 0
if req.ProposalRequest.Section != nil {
sectionIndex = req.ProposalRequest.Section.Index
}
transcriptDescription := ""
if req.ProposalRequest.Config != nil {
transcriptDescription = req.ProposalRequest.Config.TranscriptDescription
}
messages, err := req.BuildMessages(
req.ProposalRequest.WorkingTranscript,
req.ProposalRequest.Glossary,
sectionIndex,
transcriptDescription,
)
if err != nil {
return contracts.ProposalResult{}, err
}
promptMetadata, ok := prompts.LookupMetadata(req.PromptID)
if !ok {
return contracts.ProposalResult{}, fmt.Errorf("unknown prompt ID %q", req.PromptID)
}
generated, err := GenerateCandidates(ctx, Request{
ModuleKey: req.ProposalRequest.RunSpec.ModuleKey,
ModuleInstance: req.ProposalRequest.RunSpec.InstanceName,
ReplacementPolicy: req.ProposalRequest.RunSpec.ReplacementPolicy,
WorkingTranscript: req.ProposalRequest.WorkingTranscript,
Section: req.ProposalRequest.Section,
Glossary: req.ProposalRequest.Glossary,
Config: req.ProposalRequest.Config,
Messages: messages,
PromptMetadata: promptMetadata.DiagnosticsMap(),
StageName: stagename.ModuleProposal(req.ProposalRequest.RunSpec.InstanceName, sectionIndexPtr(req.ProposalRequest.Section)),
StartIndex: 0,
LLMClient: req.ProposalRequest.LLMClient,
Scheduler: req.ProposalRequest.LLMScheduler,
DiagnosticsDir: req.ProposalRequest.DiagnosticsDir,
})
if err != nil {
return contracts.ProposalResult{}, err
}
return contracts.ProposalResult{
Proposals: generated.Corrections,
Warnings: generated.Warnings,
}, nil
}

View File

@@ -0,0 +1,115 @@
package proposal_generation
import (
"context"
"errors"
"testing"
"gitea.maximumdirect.net/eric/audita/internal/core/config"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/prompts"
)
func TestExecuteModuleProposalBuildsMessagesFromSectionAndDescription(t *testing.T) {
client := &fakeStructuredClient{
responses: []StructuredCorrectionSet{
{Corrections: []StructuredCorrectionProposal{{TargetSegmentID: 1, OriginalText: "teh", CorrectedText: "the", Confidence: 0.9}}},
},
}
section := contracts.SectionMetadata{Index: 7}
cfg := config.Default()
cfg.TranscriptDescription = "Hearing transcript with role titles."
transcript := &schema.Transcript{Segments: []schema.Segment{{ID: 1, Speaker: "A", Start: 0, End: 1, Text: "teh"}}}
glossary := &schema.Glossary{Entries: []schema.GlossaryEntry{{Name: "X"}}}
var gotSectionIndex int
var gotDescription string
var gotTranscript *schema.Transcript
var gotGlossary *schema.Glossary
out, err := ExecuteModuleProposal(context.Background(), ModuleProposalRequest{
ProposalRequest: contracts.ProposalRequest{
ExecutionContext: contracts.ExecutionContext{
Config: &cfg,
WorkingTranscript: transcript,
Glossary: glossary,
Section: &section,
},
RunSpec: contracts.ModuleRunSpec{
ModuleKey: "grammar",
InstanceName: "grammar",
ReplacementPolicy: proposals.ReplacementPolicyRequireUnique,
},
LLMClient: client,
},
PromptID: prompts.PromptIDModuleGrammarProposal,
BuildMessages: func(inTranscript *schema.Transcript, inGlossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
gotSectionIndex = sectionIndex
gotDescription = transcriptDescription
gotTranscript = inTranscript
gotGlossary = inGlossary
return []contracts.LLMMessage{{Role: "system", Content: "sys"}, {Role: "user", Content: "usr"}}, nil
},
})
if err != nil {
t.Fatalf("ExecuteModuleProposal error: %v", err)
}
if gotSectionIndex != 7 {
t.Fatalf("section index: got=%d want=%d", gotSectionIndex, 7)
}
if gotDescription != cfg.TranscriptDescription {
t.Fatalf("transcript description: got=%q want=%q", gotDescription, cfg.TranscriptDescription)
}
if gotTranscript != transcript {
t.Fatalf("expected shared transcript pointer")
}
if gotGlossary != glossary {
t.Fatalf("expected shared glossary pointer")
}
if len(client.calls) != 1 || client.calls[0].StageName != "grammar:proposal:section-0007" {
t.Fatalf("unexpected stage name calls: %+v", client.calls)
}
if len(out.Proposals) != 1 || out.Proposals[0].CorrectedText != "the" {
t.Fatalf("unexpected proposals: %+v", out)
}
}
func TestExecuteModuleProposalValidatesInputs(t *testing.T) {
if _, err := ExecuteModuleProposal(context.Background(), ModuleProposalRequest{}); err == nil {
t.Fatalf("expected missing message builder error")
}
_, err := ExecuteModuleProposal(context.Background(), ModuleProposalRequest{
BuildMessages: func(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
return nil, nil
},
})
if err == nil {
t.Fatalf("expected empty prompt ID error")
}
_, err = ExecuteModuleProposal(context.Background(), ModuleProposalRequest{
PromptID: "missing.prompt.id",
BuildMessages: func(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
return []contracts.LLMMessage{{Role: "system", Content: "sys"}, {Role: "user", Content: "usr"}}, nil
},
})
if err == nil {
t.Fatalf("expected unknown prompt ID error")
}
}
func TestExecuteModuleProposalPropagatesBuilderError(t *testing.T) {
wantErr := errors.New("builder failed")
_, err := ExecuteModuleProposal(context.Background(), ModuleProposalRequest{
PromptID: prompts.PromptIDModuleGlossaryProposal,
BuildMessages: func(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
return nil, wantErr
},
})
if !errors.Is(err, wantErr) {
t.Fatalf("expected builder error, got %v", err)
}
}

View File

@@ -147,6 +147,8 @@ func skipReasonMessage(reason ProposalSkipReason) string {
return "original_text matched multiple spans under require_unique policy" return "original_text matched multiple spans under require_unique policy"
case SkipReasonNoEffect: case SkipReasonNoEffect:
return "original_text and corrected_text are identical" return "original_text and corrected_text are identical"
case SkipReasonEmptyResultingText:
return "proposal would leave the segment empty"
case SkipReasonInvalidProposal: case SkipReasonInvalidProposal:
return "proposal failed structural or policy validation" return "proposal failed structural or policy validation"
default: default:

View File

@@ -15,6 +15,7 @@ const (
SkipReasonMissingOriginalText ProposalSkipReason = "missing_original_text" SkipReasonMissingOriginalText ProposalSkipReason = "missing_original_text"
SkipReasonAmbiguousOriginal ProposalSkipReason = "ambiguous_original_text" SkipReasonAmbiguousOriginal ProposalSkipReason = "ambiguous_original_text"
SkipReasonNoEffect ProposalSkipReason = "no_effect" SkipReasonNoEffect ProposalSkipReason = "no_effect"
SkipReasonEmptyResultingText ProposalSkipReason = "empty_resulting_segment"
SkipReasonInvalidProposal ProposalSkipReason = "invalid_proposal" SkipReasonInvalidProposal ProposalSkipReason = "invalid_proposal"
) )
@@ -62,6 +63,9 @@ func PreviewProposalForSegment(segment *schema.Segment, proposal CorrectionPropo
} }
corrected := strings.Replace(segment.Text, proposal.OriginalText, proposal.CorrectedText, 1) corrected := strings.Replace(segment.Text, proposal.OriginalText, proposal.CorrectedText, 1)
if strings.TrimSpace(corrected) == "" {
return SegmentPreviewResult{SkipReason: SkipReasonEmptyResultingText}
}
return SegmentPreviewResult{ return SegmentPreviewResult{
Applicable: true, Applicable: true,
CorrectedSegmentText: corrected, CorrectedSegmentText: corrected,
@@ -70,6 +74,9 @@ func PreviewProposalForSegment(segment *schema.Segment, proposal CorrectionPropo
case ReplacementPolicyReplaceAll: case ReplacementPolicyReplaceAll:
corrected := strings.ReplaceAll(segment.Text, proposal.OriginalText, proposal.CorrectedText) corrected := strings.ReplaceAll(segment.Text, proposal.OriginalText, proposal.CorrectedText)
if strings.TrimSpace(corrected) == "" {
return SegmentPreviewResult{SkipReason: SkipReasonEmptyResultingText}
}
return SegmentPreviewResult{ return SegmentPreviewResult{
Applicable: true, Applicable: true,
CorrectedSegmentText: corrected, CorrectedSegmentText: corrected,

View File

@@ -121,6 +121,42 @@ func TestPreviewProposalForSegmentNoEffectReplacement(t *testing.T) {
} }
} }
func TestPreviewProposalForSegmentAllowsEmptyCorrectedTextWhenSegmentRemainsNonEmpty(t *testing.T) {
segment := &schema.Segment{ID: 8, Text: "uh hello"}
proposal := CorrectionProposal{
TargetSegmentID: 8,
OriginalText: "uh ",
CorrectedText: "",
Confidence: 0.9,
}
result := PreviewProposalForSegment(segment, proposal, ReplacementPolicyRequireUnique)
if !result.Applicable {
t.Fatalf("expected applicable preview, got skip reason %q", result.SkipReason)
}
if result.CorrectedSegmentText != "hello" {
t.Fatalf("unexpected corrected text: %q", result.CorrectedSegmentText)
}
}
func TestPreviewProposalForSegmentRejectsEmptyResultingSegment(t *testing.T) {
segment := &schema.Segment{ID: 8, Text: "uh"}
proposal := CorrectionProposal{
TargetSegmentID: 8,
OriginalText: "uh",
CorrectedText: "",
Confidence: 0.9,
}
result := PreviewProposalForSegment(segment, proposal, ReplacementPolicyRequireUnique)
if result.Applicable {
t.Fatal("expected non-applicable preview")
}
if result.SkipReason != SkipReasonEmptyResultingText {
t.Fatalf("expected skip reason %q, got %q", SkipReasonEmptyResultingText, result.SkipReason)
}
}
func TestPreviewProposalForSegmentPreservesInputSegment(t *testing.T) { func TestPreviewProposalForSegmentPreservesInputSegment(t *testing.T) {
segment := &schema.Segment{ID: 9, Speaker: "A", Start: 1.0, End: 2.0, Text: "rank rank"} segment := &schema.Segment{ID: 9, Speaker: "A", Start: 1.0, End: 2.0, Text: "rank rank"}
original := *segment original := *segment

View File

@@ -38,9 +38,6 @@ func (p CorrectionProposal) Validate() error {
if strings.TrimSpace(p.OriginalText) == "" { if strings.TrimSpace(p.OriginalText) == "" {
return fmt.Errorf("proposal original_text must not be empty") return fmt.Errorf("proposal original_text must not be empty")
} }
if strings.TrimSpace(p.CorrectedText) == "" {
return fmt.Errorf("proposal corrected_text must not be empty")
}
if p.Confidence < 0.0 || p.Confidence > 1.0 { if p.Confidence < 0.0 || p.Confidence > 1.0 {
return fmt.Errorf("proposal confidence must be between 0.0 and 1.0") return fmt.Errorf("proposal confidence must be between 0.0 and 1.0")
} }

View File

@@ -35,7 +35,7 @@ func TestCorrectionProposalValidate_InvalidEmptyOriginalText(t *testing.T) {
} }
} }
func TestCorrectionProposalValidate_InvalidEmptyCorrectedText(t *testing.T) { func TestCorrectionProposalValidate_AllowsEmptyCorrectedText(t *testing.T) {
proposal := CorrectionProposal{ proposal := CorrectionProposal{
TargetSegmentID: 42, TargetSegmentID: 42,
OriginalText: "gestures", OriginalText: "gestures",
@@ -43,12 +43,8 @@ func TestCorrectionProposalValidate_InvalidEmptyCorrectedText(t *testing.T) {
Confidence: 0.95, Confidence: 0.95,
} }
err := proposal.Validate() if err := proposal.Validate(); err != nil {
if err == nil { t.Fatalf("expected empty corrected_text to be allowed, got %v", err)
t.Fatal("expected validation error, got nil")
}
if err.Error() != "proposal corrected_text must not be empty" {
t.Fatalf("unexpected error: %v", err)
} }
} }

View File

@@ -5,6 +5,7 @@ import (
"encoding/hex" "encoding/hex"
"encoding/json" "encoding/json"
"fmt" "fmt"
"sort"
"strings" "strings"
) )
@@ -28,6 +29,15 @@ type Schema struct {
SHA256 string `json:"sha256"` SHA256 string `json:"sha256"`
} }
func (s Schema) DiagnosticsMap() map[string]any {
return map[string]any{
"id": s.ID,
"version": s.Version,
"name": s.Name,
"sha256": s.SHA256,
}
}
var registry = map[Key]Schema{ var registry = map[Key]Schema{
CorrectionSetKey: mustBuildSchema( CorrectionSetKey: mustBuildSchema(
correctionSetSchemaID, correctionSetSchemaID,
@@ -43,6 +53,20 @@ var registry = map[Key]Schema{
), ),
} }
func Registered() []Schema {
keys := make([]string, 0, len(registry))
for key := range registry {
keys = append(keys, string(key))
}
sort.Strings(keys)
out := make([]Schema, 0, len(keys))
for _, key := range keys {
out = append(out, cloneSchema(registry[Key(key)]))
}
return out
}
// Lookup returns a copy of the registered schema for the provided key. // Lookup returns a copy of the registered schema for the provided key.
func Lookup(key Key) (Schema, bool) { func Lookup(key Key) (Schema, bool) {
schema, ok := registry[key] schema, ok := registry[key]

View File

@@ -89,3 +89,20 @@ func TestLookupReturnsSchemaCopy(t *testing.T) {
t.Fatalf("expected lookup to return independent schema copy") t.Fatalf("expected lookup to return independent schema copy")
} }
} }
func TestDiagnosticsMapIncludesStableSchemaMetadataShapeForAllSchemas(t *testing.T) {
registered := Registered()
if len(registered) == 0 {
t.Fatalf("expected registered response schemas")
}
for _, schema := range registered {
metadataMap := schema.DiagnosticsMap()
if metadataMap["id"] != schema.ID ||
metadataMap["version"] != schema.Version ||
metadataMap["name"] != schema.Name ||
metadataMap["sha256"] != schema.SHA256 {
t.Fatalf("unexpected diagnostics metadata map for %q: %+v", schema.ID, metadataMap)
}
}
}

View File

@@ -15,6 +15,7 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/framework/llm" "gitea.maximumdirect.net/eric/audita/internal/framework/llm"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals" "gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/validators" "gitea.maximumdirect.net/eric/audita/internal/framework/validators"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
validatormetadata "gitea.maximumdirect.net/eric/audita/internal/validators/metadata" validatormetadata "gitea.maximumdirect.net/eric/audita/internal/validators/metadata"
) )
@@ -37,18 +38,19 @@ type ValidationScheduler = contracts.LLMScheduler
// ModuleResult captures deterministic per-module execution output. // ModuleResult captures deterministic per-module execution output.
type ModuleResult struct { type ModuleResult struct {
ModuleKey string `json:"module_key"` ModuleKey string `json:"module_key"`
ModuleInstance string `json:"module_instance"` ModuleInstance string `json:"module_instance"`
ReplacementPolicy proposals.ReplacementPolicy `json:"replacement_policy"` ReplacementPolicy proposals.ReplacementPolicy `json:"replacement_policy"`
Status string `json:"status"` Status string `json:"status"`
ProposalCount int `json:"proposal_count"` ProposalCount int `json:"proposal_count"`
ValidatorDecisions []ValidatorDecisionRecord `json:"validator_decisions,omitempty"` Warnings []stagewarnings.StageWarning `json:"warnings,omitempty"`
ValidatorRejected []ValidatorRejectedChange `json:"validator_rejected,omitempty"` ValidatorDecisions []ValidatorDecisionRecord `json:"validator_decisions,omitempty"`
AppliedChanges []proposals.AppliedChange `json:"applied_changes,omitempty"` ValidatorRejected []ValidatorRejectedChange `json:"validator_rejected,omitempty"`
SkippedChanges []proposals.SkippedChange `json:"skipped_changes,omitempty"` AppliedChanges []proposals.AppliedChange `json:"applied_changes,omitempty"`
ErrorMessage string `json:"error_message,omitempty"` SkippedChanges []proposals.SkippedChange `json:"skipped_changes,omitempty"`
StartedAt time.Time `json:"started_at"` ErrorMessage string `json:"error_message,omitempty"`
CompletedAt time.Time `json:"completed_at"` StartedAt time.Time `json:"started_at"`
CompletedAt time.Time `json:"completed_at"`
} }
type ValidatorDecisionRecord struct { type ValidatorDecisionRecord struct {
@@ -176,6 +178,7 @@ func (r *Runner) Run(ctx context.Context, input RunInput) (RunOutput, error) {
ReplacementPolicy: policy, ReplacementPolicy: policy,
Status: ModuleStatusFailed, Status: ModuleStatusFailed,
ProposalCount: pipelineResult.ProposalCount, ProposalCount: pipelineResult.ProposalCount,
Warnings: pipelineResult.Warnings,
ValidatorDecisions: pipelineResult.ValidatorDecisions, ValidatorDecisions: pipelineResult.ValidatorDecisions,
ValidatorRejected: pipelineResult.ValidatorRejected, ValidatorRejected: pipelineResult.ValidatorRejected,
ErrorMessage: pipelineErr.Error(), ErrorMessage: pipelineErr.Error(),
@@ -199,6 +202,7 @@ func (r *Runner) Run(ctx context.Context, input RunInput) (RunOutput, error) {
ReplacementPolicy: policy, ReplacementPolicy: policy,
Status: ModuleStatusSuccess, Status: ModuleStatusSuccess,
ProposalCount: pipelineResult.ProposalCount, ProposalCount: pipelineResult.ProposalCount,
Warnings: pipelineResult.Warnings,
ValidatorDecisions: pipelineResult.ValidatorDecisions, ValidatorDecisions: pipelineResult.ValidatorDecisions,
ValidatorRejected: pipelineResult.ValidatorRejected, ValidatorRejected: pipelineResult.ValidatorRejected,
AppliedChanges: applyResult.Applied, AppliedChanges: applyResult.Applied,
@@ -232,12 +236,14 @@ type collectSectionProposalsInput struct {
type sectionProposals struct { type sectionProposals struct {
meta contracts.SectionMetadata meta contracts.SectionMetadata
corrected []proposals.CorrectionProposal corrected []proposals.CorrectionProposal
warnings []stagewarnings.StageWarning
} }
type sectionProposalResult struct { type sectionProposalResult struct {
sectionPos int sectionPos int
section chunking.Section section chunking.Section
corrected []proposals.CorrectionProposal corrected []proposals.CorrectionProposal
warnings []stagewarnings.StageWarning
err error err error
} }
@@ -247,12 +253,14 @@ type sectionValidationResult struct {
approved []proposals.EnrichedCorrectionProposal approved []proposals.EnrichedCorrectionProposal
decisions []ValidatorDecisionRecord decisions []ValidatorDecisionRecord
rejected []ValidatorRejectedChange rejected []ValidatorRejectedChange
warnings []stagewarnings.StageWarning
err error err error
} }
type modulePipelineResult struct { type modulePipelineResult struct {
ProposalCount int ProposalCount int
Approved []proposals.EnrichedCorrectionProposal Approved []proposals.EnrichedCorrectionProposal
Warnings []stagewarnings.StageWarning
ValidatorDecisions []ValidatorDecisionRecord ValidatorDecisions []ValidatorDecisionRecord
ValidatorRejected []ValidatorRejectedChange ValidatorRejected []ValidatorRejectedChange
} }
@@ -271,7 +279,7 @@ func collectSectionProposals(ctx context.Context, input collectSectionProposalsI
go func() { go func() {
defer wg.Done() defer wg.Done()
meta := contracts.SectionMetadataFromSection(section) meta := contracts.SectionMetadataFromSection(section)
corrected, err := input.Module.Propose(withModuleInstanceContext(runCtx, input.Spec.InstanceName), contracts.ProposalRequest{ proposalResult, err := input.Module.Propose(withModuleInstanceContext(runCtx, input.Spec.InstanceName), contracts.ProposalRequest{
ExecutionContext: contracts.ExecutionContext{ ExecutionContext: contracts.ExecutionContext{
Config: input.Config, Config: input.Config,
WorkingTranscript: transcriptFromSection(section), WorkingTranscript: transcriptFromSection(section),
@@ -291,7 +299,8 @@ func collectSectionProposals(ctx context.Context, input collectSectionProposalsI
case results <- sectionProposalResult{ case results <- sectionProposalResult{
sectionPos: sectionPos, sectionPos: sectionPos,
section: section, section: section,
corrected: corrected, corrected: proposalResult.Proposals,
warnings: proposalResult.Warnings,
err: err, err: err,
}: }:
case <-runCtx.Done(): case <-runCtx.Done():
@@ -308,6 +317,7 @@ func collectSectionProposals(ctx context.Context, input collectSectionProposalsI
func runModulePipeline(ctx context.Context, input collectSectionProposalsInput) (modulePipelineResult, error) { func runModulePipeline(ctx context.Context, input collectSectionProposalsInput) (modulePipelineResult, error) {
out := modulePipelineResult{ out := modulePipelineResult{
Approved: make([]proposals.EnrichedCorrectionProposal, 0), Approved: make([]proposals.EnrichedCorrectionProposal, 0),
Warnings: make([]stagewarnings.StageWarning, 0),
ValidatorDecisions: make([]ValidatorDecisionRecord, 0), ValidatorDecisions: make([]ValidatorDecisionRecord, 0),
ValidatorRejected: make([]ValidatorRejectedChange, 0), ValidatorRejected: make([]ValidatorRejectedChange, 0),
} }
@@ -352,6 +362,7 @@ func runModulePipeline(ctx context.Context, input collectSectionProposalsInput)
if firstErr != nil { if firstErr != nil {
continue continue
} }
out.Warnings = append(out.Warnings, result.warnings...)
pending[result.sectionPos] = result pending[result.sectionPos] = result
for { for {
@@ -401,6 +412,7 @@ func runModulePipeline(ctx context.Context, input collectSectionProposalsInput)
approved: validated.approved, approved: validated.approved,
decisions: validated.decisions, decisions: validated.decisions,
rejected: validated.rejected, rejected: validated.rejected,
warnings: validated.warnings,
err: err, err: err,
} }
}(nextSectionToProcess, sectionEnriched, sectionMeta) }(nextSectionToProcess, sectionEnriched, sectionMeta)
@@ -424,6 +436,7 @@ func runModulePipeline(ctx context.Context, input collectSectionProposalsInput)
break break
} }
out.Approved = append(out.Approved, res.approved...) out.Approved = append(out.Approved, res.approved...)
out.Warnings = append(out.Warnings, res.warnings...)
out.ValidatorDecisions = append(out.ValidatorDecisions, res.decisions...) out.ValidatorDecisions = append(out.ValidatorDecisions, res.decisions...)
out.ValidatorRejected = append(out.ValidatorRejected, res.rejected...) out.ValidatorRejected = append(out.ValidatorRejected, res.rejected...)
} }
@@ -440,6 +453,38 @@ func runModulePipeline(ctx context.Context, input collectSectionProposalsInput)
} }
return validatorOrder[out.ValidatorRejected[i].ValidatorName] < validatorOrder[out.ValidatorRejected[j].ValidatorName] return validatorOrder[out.ValidatorRejected[i].ValidatorName] < validatorOrder[out.ValidatorRejected[j].ValidatorName]
}) })
sort.SliceStable(out.Warnings, func(i, j int) bool {
leftSection, rightSection := -1, -1
if out.Warnings[i].SectionIndex != nil {
leftSection = *out.Warnings[i].SectionIndex
}
if out.Warnings[j].SectionIndex != nil {
rightSection = *out.Warnings[j].SectionIndex
}
if leftSection != rightSection {
return leftSection < rightSection
}
leftBatch, rightBatch := -1, -1
if out.Warnings[i].BatchIndex != nil {
leftBatch = *out.Warnings[i].BatchIndex
}
if out.Warnings[j].BatchIndex != nil {
rightBatch = *out.Warnings[j].BatchIndex
}
if leftBatch != rightBatch {
return leftBatch < rightBatch
}
if out.Warnings[i].ValidatorName != out.Warnings[j].ValidatorName {
return validatorOrder[out.Warnings[i].ValidatorName] < validatorOrder[out.Warnings[j].ValidatorName]
}
if out.Warnings[i].Scope != out.Warnings[j].Scope {
return out.Warnings[i].Scope < out.Warnings[j].Scope
}
if out.Warnings[i].ReasonCode != out.Warnings[j].ReasonCode {
return out.Warnings[i].ReasonCode < out.Warnings[j].ReasonCode
}
return out.Warnings[i].Message < out.Warnings[j].Message
})
if firstErr != nil { if firstErr != nil {
return out, firstErr return out, firstErr
@@ -467,11 +512,13 @@ type validateSectionCandidatesResult struct {
approved []proposals.EnrichedCorrectionProposal approved []proposals.EnrichedCorrectionProposal
decisions []ValidatorDecisionRecord decisions []ValidatorDecisionRecord
rejected []ValidatorRejectedChange rejected []ValidatorRejectedChange
warnings []stagewarnings.StageWarning
} }
func validateSectionCandidates(ctx context.Context, input validateSectionCandidatesInput) (validateSectionCandidatesResult, error) { func validateSectionCandidates(ctx context.Context, input validateSectionCandidatesInput) (validateSectionCandidatesResult, error) {
decisions := make([]ValidatorDecisionRecord, 0) decisions := make([]ValidatorDecisionRecord, 0)
rejected := make([]ValidatorRejectedChange, 0) rejected := make([]ValidatorRejectedChange, 0)
warnings := make([]stagewarnings.StageWarning, 0)
eligible := append([]proposals.EnrichedCorrectionProposal(nil), input.SectionEnriched...) eligible := append([]proposals.EnrichedCorrectionProposal(nil), input.SectionEnriched...)
for _, validator := range input.Validators { for _, validator := range input.Validators {
@@ -480,7 +527,7 @@ func validateSectionCandidates(ctx context.Context, input validateSectionCandida
diagnosticsWriter = &llmDiagnosticsWriterAdapter{ diagnosticsWriter = &llmDiagnosticsWriterAdapter{
writer: llm.NewDiagnosticsWriter( writer: llm.NewDiagnosticsWriter(
filepath.Join(input.DiagnosticsDir, input.Spec.InstanceName), filepath.Join(input.DiagnosticsDir, input.Spec.InstanceName),
validatorSecrets(input.Config), llm.ConfiguredSecrets(input.Config),
), ),
} }
} }
@@ -512,6 +559,7 @@ func validateSectionCandidates(ctx context.Context, input validateSectionCandida
approved: eligible, approved: eligible,
decisions: decisions, decisions: decisions,
rejected: rejected, rejected: rejected,
warnings: warnings,
}, fmt.Errorf("validator %q failed: %w", validator.Name(), err) }, fmt.Errorf("validator %q failed: %w", validator.Name(), err)
} }
if err := validators.EnforceDecisionCardinality(eligible, vResult.Decisions); err != nil { if err := validators.EnforceDecisionCardinality(eligible, vResult.Decisions); err != nil {
@@ -519,8 +567,10 @@ func validateSectionCandidates(ctx context.Context, input validateSectionCandida
approved: eligible, approved: eligible,
decisions: decisions, decisions: decisions,
rejected: rejected, rejected: rejected,
warnings: warnings,
}, fmt.Errorf("validator %q cardinality failed: %w", validator.Name(), err) }, fmt.Errorf("validator %q cardinality failed: %w", validator.Name(), err)
} }
warnings = append(warnings, vResult.Warnings...)
nextEligible := make([]proposals.EnrichedCorrectionProposal, 0, len(eligible)) nextEligible := make([]proposals.EnrichedCorrectionProposal, 0, len(eligible))
byIndex := make(map[int]proposals.EnrichedCorrectionProposal, len(eligible)) byIndex := make(map[int]proposals.EnrichedCorrectionProposal, len(eligible))
@@ -562,6 +612,7 @@ func validateSectionCandidates(ctx context.Context, input validateSectionCandida
approved: eligible, approved: eligible,
decisions: decisions, decisions: decisions,
rejected: rejected, rejected: rejected,
warnings: warnings,
}, nil }, nil
} }
@@ -693,15 +744,3 @@ func (a *llmDiagnosticsWriterAdapter) WriteInteraction(stage string, requestMeta
ErrorPayloadPath: art.ErrorPayloadPath, ErrorPayloadPath: art.ErrorPayloadPath,
}, nil }, nil
} }
func validatorSecrets(cfg *config.Config) []string {
if cfg == nil {
return nil
}
effective := cfg.EffectiveValidationLLMConfig()
return []string{
cfg.PrimaryLLM.APIKey,
effective.APIKey,
cfg.ValidationLLM.APIKey,
}
}

View File

@@ -47,11 +47,12 @@ type fakeModule struct {
func (m fakeModule) Key() string { return m.key } func (m fakeModule) Key() string { return m.key }
func (m fakeModule) ReplacementPolicy() proposals.ReplacementPolicy { return m.policy } func (m fakeModule) ReplacementPolicy() proposals.ReplacementPolicy { return m.policy }
func (m fakeModule) Validators() []contracts.Validator { return m.validators } func (m fakeModule) Validators() []contracts.Validator { return m.validators }
func (m fakeModule) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { func (m fakeModule) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
if m.proposeF == nil { if m.proposeF == nil {
return nil, nil return contracts.ProposalResult{}, nil
} }
return m.proposeF(req) proposalsOut, err := m.proposeF(req)
return contracts.ProposalResult{Proposals: proposalsOut}, err
} }
type fakeValidator struct { type fakeValidator struct {
@@ -1191,7 +1192,7 @@ func TestRunnerLLMValidatorRejectionPreventsApplication(t *testing.T) {
} }
} }
func TestRunnerLLMValidatorMalformedResponseFailsWithPartialProgress(t *testing.T) { func TestRunnerLLMValidatorMalformedResponseRejectsBatchAndKeepsPartialProgress(t *testing.T) {
client := &fakeStructuredClient{responses: []validators.LLMValidationResponse{{Validations: []validators.LLMValidationDecision{{CorrectionIndex: 99, Approved: true, Confidence: 0.9, Reason: "bad index"}}}}} client := &fakeStructuredClient{responses: []validators.LLMValidationResponse{{Validations: []validators.LLMValidationDecision{{CorrectionIndex: 99, Approved: true, Confidence: 0.9, Reason: "bad index"}}}}}
llmValidator, _ := validators.NewLLMBackedValidator("spoken_form_plausibility_review", validators.LLMValidatorTypeSpokenFormPlausibility, "") llmValidator, _ := validators.NewLLMBackedValidator("spoken_form_plausibility_review", validators.LLMValidatorTypeSpokenFormPlausibility, "")
r := New(fakeFactory{modules: map[string]contracts.TranscriptModule{ r := New(fakeFactory{modules: map[string]contracts.TranscriptModule{
@@ -1209,15 +1210,21 @@ func TestRunnerLLMValidatorMalformedResponseFailsWithPartialProgress(t *testing.
ModuleSpecs: []contracts.ModuleRunSpec{{ModuleKey: "m1", InstanceName: "m1"}, {ModuleKey: "m2", InstanceName: "m2"}}, ModuleSpecs: []contracts.ModuleRunSpec{{ModuleKey: "m1", InstanceName: "m1"}, {ModuleKey: "m2", InstanceName: "m2"}},
ValidationLLMClient: client, ValidationLLMClient: client,
}) })
if err == nil { if err != nil {
t.Fatal("expected llm validator failure") t.Fatalf("expected malformed validator response to downgrade, got %v", err)
} }
if out.FinalTranscript.Segments[0].Text != "the cat" { if out.FinalTranscript.Segments[0].Text != "the cat" {
t.Fatalf("expected partial progress retained") t.Fatalf("expected partial progress retained")
} }
if len(out.ModuleResults) != 2 || len(out.ModuleResults[1].ValidatorRejected) != 1 {
t.Fatalf("expected second module rejection, got %+v", out.ModuleResults)
}
if len(out.ModuleResults[1].Warnings) != 1 || out.ModuleResults[1].Warnings[0].ReasonCode != validators.ReasonValidatorMalformed {
t.Fatalf("expected malformed warning, got %+v", out.ModuleResults[1].Warnings)
}
} }
func TestRunnerLLMValidatorMissingDecisionFails(t *testing.T) { func TestRunnerLLMValidatorMissingDecisionRejectsBatch(t *testing.T) {
client := &fakeStructuredClient{responses: []validators.LLMValidationResponse{{Validations: []validators.LLMValidationDecision{}}}} client := &fakeStructuredClient{responses: []validators.LLMValidationResponse{{Validations: []validators.LLMValidationDecision{}}}}
llmValidator, _ := validators.NewLLMBackedValidator("spoken_form_plausibility_review", validators.LLMValidatorTypeSpokenFormPlausibility, "") llmValidator, _ := validators.NewLLMBackedValidator("spoken_form_plausibility_review", validators.LLMValidatorTypeSpokenFormPlausibility, "")
r := New(fakeFactory{modules: map[string]contracts.TranscriptModule{ r := New(fakeFactory{modules: map[string]contracts.TranscriptModule{
@@ -1232,12 +1239,12 @@ func TestRunnerLLMValidatorMissingDecisionFails(t *testing.T) {
ModuleSpecs: []contracts.ModuleRunSpec{{ModuleKey: "m", InstanceName: "m"}}, ModuleSpecs: []contracts.ModuleRunSpec{{ModuleKey: "m", InstanceName: "m"}},
ValidationLLMClient: client, ValidationLLMClient: client,
}) })
if err == nil { if err != nil {
t.Fatal("expected missing decision failure") t.Fatalf("expected missing decision downgrade, got %v", err)
} }
} }
func TestRunnerLLMValidatorDuplicateDecisionFails(t *testing.T) { func TestRunnerLLMValidatorDuplicateDecisionRejectsBatch(t *testing.T) {
client := &fakeStructuredClient{responses: []validators.LLMValidationResponse{{Validations: []validators.LLMValidationDecision{ client := &fakeStructuredClient{responses: []validators.LLMValidationResponse{{Validations: []validators.LLMValidationDecision{
{CorrectionIndex: 0, Approved: true, Confidence: 0.9, Reason: "ok"}, {CorrectionIndex: 0, Approved: true, Confidence: 0.9, Reason: "ok"},
{CorrectionIndex: 0, Approved: false, Confidence: 0.9, Reason: "dup"}, {CorrectionIndex: 0, Approved: false, Confidence: 0.9, Reason: "dup"},
@@ -1255,8 +1262,8 @@ func TestRunnerLLMValidatorDuplicateDecisionFails(t *testing.T) {
ModuleSpecs: []contracts.ModuleRunSpec{{ModuleKey: "m", InstanceName: "m"}}, ModuleSpecs: []contracts.ModuleRunSpec{{ModuleKey: "m", InstanceName: "m"}},
ValidationLLMClient: client, ValidationLLMClient: client,
}) })
if err == nil { if err != nil {
t.Fatal("expected duplicate decision failure") t.Fatalf("expected duplicate decision downgrade, got %v", err)
} }
} }
@@ -1300,12 +1307,13 @@ func TestRunnerLLMValidatorBatchingAndSchedulerUsage(t *testing.T) {
} }
func TestRunnerLLMValidatorDiagnosticsWrittenAndRedacted(t *testing.T) { func TestRunnerLLMValidatorDiagnosticsWrittenAndRedacted(t *testing.T) {
secret := "super-secret-key" primarySecret := "runner-primary-secret"
client := &fakeStructuredClient{responses: []validators.LLMValidationResponse{{Validations: []validators.LLMValidationDecision{{CorrectionIndex: 0, Approved: true, Confidence: 0.9, Reason: secret}}}}} validationSecret := "runner-validation-secret"
client := &fakeStructuredClient{responses: []validators.LLMValidationResponse{{Validations: []validators.LLMValidationDecision{{CorrectionIndex: 0, Approved: true, Confidence: 0.9, Reason: validationSecret}}}}}
llmValidator, _ := validators.NewLLMBackedValidator("spoken_form_plausibility_review", validators.LLMValidatorTypeSpokenFormPlausibility, "") llmValidator, _ := validators.NewLLMBackedValidator("spoken_form_plausibility_review", validators.LLMValidatorTypeSpokenFormPlausibility, "")
cfg := config.Default() cfg := config.Default()
cfg.PrimaryLLM.APIKey = secret cfg.PrimaryLLM.APIKey = primarySecret
cfg.ValidationLLM.APIKey = secret cfg.ValidationLLM.APIKey = validationSecret
diagDir := t.TempDir() diagDir := t.TempDir()
r := New(fakeFactory{modules: map[string]contracts.TranscriptModule{ r := New(fakeFactory{modules: map[string]contracts.TranscriptModule{
"m": fakeModule{key: "m", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{llmValidator}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { "m": fakeModule{key: "m", policy: proposals.ReplacementPolicyRequireUnique, validators: []contracts.Validator{llmValidator}, proposeF: func(req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) {
@@ -1314,7 +1322,7 @@ func TestRunnerLLMValidatorDiagnosticsWrittenAndRedacted(t *testing.T) {
}}) }})
out, err := r.Run(context.Background(), RunInput{ out, err := r.Run(context.Background(), RunInput{
Config: &cfg, Config: &cfg,
Transcript: &schema.Transcript{Segments: []schema.Segment{{ID: 1, Text: "There were gestures at the temple."}}}, Transcript: &schema.Transcript{Segments: []schema.Segment{{ID: 1, Text: "There were gestures at the temple. " + primarySecret}}},
ModuleSpecs: []contracts.ModuleRunSpec{{ModuleKey: "m", InstanceName: "m"}}, ModuleSpecs: []contracts.ModuleRunSpec{{ModuleKey: "m", InstanceName: "m"}},
ValidationLLMClient: client, ValidationLLMClient: client,
ValidationDiagnosticsDir: diagDir, ValidationDiagnosticsDir: diagDir,
@@ -1325,12 +1333,26 @@ func TestRunnerLLMValidatorDiagnosticsWrittenAndRedacted(t *testing.T) {
if len(out.ModuleResults[0].ValidatorDecisions) == 0 || out.ModuleResults[0].ValidatorDecisions[0].DiagnosticArtifactPath == "" { if len(out.ModuleResults[0].ValidatorDecisions) == 0 || out.ModuleResults[0].ValidatorDecisions[0].DiagnosticArtifactPath == "" {
t.Fatalf("expected diagnostic artifact path on decision") t.Fatalf("expected diagnostic artifact path on decision")
} }
matches, globErr := filepath.Glob(filepath.Join(diagDir, "m", "*.json"))
if globErr != nil {
t.Fatalf("glob diagnostics: %v", globErr)
}
if len(matches) == 0 {
t.Fatalf("expected diagnostics JSON artifacts under %s", filepath.Join(diagDir, "m"))
}
for _, path := range matches {
raw, readErr := os.ReadFile(path)
if readErr != nil {
t.Fatalf("read diagnostic %q: %v", path, readErr)
}
if strings.Contains(string(raw), primarySecret) || strings.Contains(string(raw), validationSecret) {
t.Fatalf("configured secret leaked in diagnostics %q: %s", path, string(raw))
}
}
raw, readErr := os.ReadFile(out.ModuleResults[0].ValidatorDecisions[0].DiagnosticArtifactPath) raw, readErr := os.ReadFile(out.ModuleResults[0].ValidatorDecisions[0].DiagnosticArtifactPath)
if readErr != nil { if readErr != nil {
t.Fatalf("read diagnostic: %v", readErr) t.Fatalf("read decision diagnostic: %v", readErr)
}
if strings.Contains(string(raw), secret) {
t.Fatalf("secret leaked in diagnostics: %s", string(raw))
} }
if !strings.Contains(string(raw), "[REDACTED]") { if !strings.Contains(string(raw), "[REDACTED]") {
t.Fatalf("expected redaction marker in diagnostics") t.Fatalf("expected redaction marker in diagnostics")
@@ -1355,7 +1377,7 @@ type proposalGenerationModule struct {
func (m proposalGenerationModule) Key() string { return m.key } func (m proposalGenerationModule) Key() string { return m.key }
func (m proposalGenerationModule) ReplacementPolicy() proposals.ReplacementPolicy { return m.policy } func (m proposalGenerationModule) ReplacementPolicy() proposals.ReplacementPolicy { return m.policy }
func (m proposalGenerationModule) Validators() []contracts.Validator { return nil } func (m proposalGenerationModule) Validators() []contracts.Validator { return nil }
func (m proposalGenerationModule) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { func (m proposalGenerationModule) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
result, err := proposal_generation.GenerateCandidates(ctx, proposal_generation.Request{ result, err := proposal_generation.GenerateCandidates(ctx, proposal_generation.Request{
ModuleKey: req.RunSpec.ModuleKey, ModuleKey: req.RunSpec.ModuleKey,
ModuleInstance: req.RunSpec.InstanceName, ModuleInstance: req.RunSpec.InstanceName,
@@ -1372,9 +1394,9 @@ func (m proposalGenerationModule) Propose(ctx context.Context, req contracts.Pro
DiagnosticsDir: req.DiagnosticsDir, DiagnosticsDir: req.DiagnosticsDir,
}) })
if err != nil { if err != nil {
return nil, err return contracts.ProposalResult{}, err
} }
return result.Corrections, nil return contracts.ProposalResult{Proposals: result.Corrections, Warnings: result.Warnings}, nil
} }
type fakeProposalStructuredClient struct { type fakeProposalStructuredClient struct {

View File

@@ -0,0 +1,24 @@
package stagename
import (
"fmt"
)
func ModuleProposal(moduleInstance string, sectionIndex *int) string {
if sectionIndex == nil || *sectionIndex == 0 {
return fmt.Sprintf("%s:proposal", moduleInstance)
}
return fmt.Sprintf("%s:proposal:section-%04d", moduleInstance, *sectionIndex)
}
func ProposalGeneration(moduleInstance string, sectionIndex *int) string {
base := fmt.Sprintf("%s:proposal-generation", moduleInstance)
if sectionIndex == nil {
return base
}
return fmt.Sprintf("%s:section-%04d", base, *sectionIndex)
}
func ValidatorBatch(moduleInstance string, validatorName string, batchIndex int) string {
return fmt.Sprintf("%s:%s:batch-%04d", moduleInstance, validatorName, batchIndex)
}

View File

@@ -0,0 +1,33 @@
package stagename
import (
"testing"
)
func TestModuleProposalStageName(t *testing.T) {
if got := ModuleProposal("grammar", nil); got != "grammar:proposal" {
t.Fatalf("unexpected stage name without section: %q", got)
}
sectionIndex := 7
if got := ModuleProposal("grammar", &sectionIndex); got != "grammar:proposal:section-0007" {
t.Fatalf("unexpected stage name with section: %q", got)
}
}
func TestProposalGenerationStageName(t *testing.T) {
if got := ProposalGeneration("grammar", nil); got != "grammar:proposal-generation" {
t.Fatalf("unexpected proposal generation stage name without section: %q", got)
}
sectionIndex := 3
if got := ProposalGeneration("grammar", &sectionIndex); got != "grammar:proposal-generation:section-0003" {
t.Fatalf("unexpected proposal generation stage name with section: %q", got)
}
}
func TestValidatorBatchStageName(t *testing.T) {
if got := ValidatorBatch("homophones_1", "spoken_form_plausibility_review", 12); got != "homophones_1:spoken_form_plausibility_review:batch-0012" {
t.Fatalf("unexpected validator batch stage name: %q", got)
}
}

View File

@@ -0,0 +1,28 @@
package structuredoutput
import "strings"
var malformedMarkers = []string{
"malformed structured output",
"decode structured output:",
"decode provider response envelope:",
"provider response missing choices",
"provider response missing assistant message content",
"provider response assistant message content is empty",
"provider response assistant message content is not valid JSON",
}
// IsMalformedError reports whether err matches provider malformed
// structured-output failure markers that should be downgraded.
func IsMalformedError(err error) bool {
if err == nil {
return false
}
msg := err.Error()
for _, marker := range malformedMarkers {
if strings.Contains(msg, marker) {
return true
}
}
return false
}

View File

@@ -0,0 +1,32 @@
package structuredoutput
import (
"errors"
"testing"
)
func TestIsMalformedError(t *testing.T) {
cases := []struct {
name string
err error
want bool
}{
{name: "nil", err: nil, want: false},
{name: "generic", err: errors.New("network timeout"), want: false},
{name: "malformed", err: errors.New("malformed structured output"), want: true},
{name: "decode structured", err: errors.New("decode structured output: unexpected end of JSON input"), want: true},
{name: "missing choices", err: errors.New("provider response missing choices"), want: true},
{name: "missing content", err: errors.New("provider response missing assistant message content"), want: true},
{name: "empty content", err: errors.New("provider response assistant message content is empty"), want: true},
{name: "invalid content json", err: errors.New("provider response assistant message content is not valid JSON"), want: true},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
got := IsMalformedError(tc.err)
if got != tc.want {
t.Fatalf("IsMalformedError(%v): got=%v want=%v", tc.err, got, tc.want)
}
})
}
}

View File

@@ -4,8 +4,35 @@ import (
"context" "context"
"fmt" "fmt"
"strings" "strings"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
) )
type ProposalShapeValidator struct{}
func (v ProposalShapeValidator) Name() string { return "proposal_shape" }
func (v ProposalShapeValidator) Validate(_ context.Context, req Request) (Result, error) {
decisions := make([]Decision, 0, len(req.CandidateProposal))
for _, c := range req.CandidateProposal {
switch {
case c.TargetSegmentID <= 0:
decisions = append(decisions, rejection(c.ProposalIndex, ReasonInvalidTargetSegment, "proposal target segment id must be positive"))
case strings.TrimSpace(c.OriginalText) == "":
decisions = append(decisions, rejection(c.ProposalIndex, ReasonEmptyOriginalText, "proposal original_text must not be empty"))
case c.Confidence < 0.0 || c.Confidence > 1.0:
decisions = append(decisions, rejection(c.ProposalIndex, ReasonInvalidConfidence, "proposal confidence must be between 0.0 and 1.0"))
default:
decisions = append(decisions, approval(c.ProposalIndex))
}
}
if err := EnforceDecisionCardinality(req.CandidateProposal, decisions); err != nil {
return Result{}, err
}
return Result{ValidatorName: v.Name(), Decisions: decisions}, nil
}
type ConfidenceThresholdValidator struct{} type ConfidenceThresholdValidator struct{}
func (v ConfidenceThresholdValidator) Name() string { return "confidence_threshold" } func (v ConfidenceThresholdValidator) Name() string { return "confidence_threshold" }
@@ -62,10 +89,23 @@ type NonEmptyCorrectionValidator struct{}
func (v NonEmptyCorrectionValidator) Name() string { return "non_empty_corrected_text" } func (v NonEmptyCorrectionValidator) Name() string { return "non_empty_corrected_text" }
func (v NonEmptyCorrectionValidator) Validate(_ context.Context, req Request) (Result, error) { func (v NonEmptyCorrectionValidator) Validate(_ context.Context, req Request) (Result, error) {
segmentsByID := make(map[int]schema.Segment)
if req.WorkingTranscript != nil {
for _, seg := range req.WorkingTranscript.Segments {
segmentsByID[seg.ID] = seg
}
}
decisions := make([]Decision, 0, len(req.CandidateProposal)) decisions := make([]Decision, 0, len(req.CandidateProposal))
for _, c := range req.CandidateProposal { for _, c := range req.CandidateProposal {
if strings.TrimSpace(c.CorrectedText) == "" { segment, ok := segmentsByID[c.TargetSegmentID]
decisions = append(decisions, rejection(c.ProposalIndex, ReasonEmptyCorrectedText, "corrected_text must not be empty")) if !ok {
decisions = append(decisions, approval(c.ProposalIndex))
continue
}
preview := proposals.PreviewProposalForSegment(&segment, c.CorrectionProposal, req.ReplacementPolicy)
if preview.SkipReason == proposals.SkipReasonEmptyResultingText {
decisions = append(decisions, rejection(c.ProposalIndex, ReasonEmptyResultingText, "proposal would leave the segment empty"))
continue continue
} }
decisions = append(decisions, approval(c.ProposalIndex)) decisions = append(decisions, approval(c.ProposalIndex))

View File

@@ -11,6 +11,9 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/core/schema" "gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals" "gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/framework/responseschema" "gitea.maximumdirect.net/eric/audita/internal/framework/responseschema"
"gitea.maximumdirect.net/eric/audita/internal/framework/stagename"
"gitea.maximumdirect.net/eric/audita/internal/framework/structuredoutput"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
"gitea.maximumdirect.net/eric/audita/internal/prompts" "gitea.maximumdirect.net/eric/audita/internal/prompts"
) )
@@ -78,7 +81,20 @@ func (v *LLMBackedValidator) Validate(ctx context.Context, req Request) (Result,
maxTokens = req.Config.ValidationMaxPromptTokens maxTokens = req.Config.ValidationMaxPromptTokens
} }
batches, err := ChunkLLMValidationItems(validationReq.Items, maxTokens, v.estimator) warnings := append([]stagewarnings.StageWarning(nil), oversizedValidationWarnings(v.name, maxTokens, validationReq.Items, v.estimator)...)
oversized := oversizedValidationDecisions(validationReq.Items, maxTokens, v.estimator)
itemsForBatching := filterItemsByDecision(validationReq.Items, oversized)
if len(itemsForBatching) == 0 {
all := append([]Decision(nil), immediate...)
all = append(all, oversized...)
if err := EnforceDecisionCardinality(req.CandidateProposal, all); err != nil {
return Result{}, err
}
sort.SliceStable(all, func(i, j int) bool { return all[i].ProposalIndex < all[j].ProposalIndex })
return Result{ValidatorName: v.name, Decisions: all, Warnings: warnings}, nil
}
batches, err := ChunkLLMValidationItems(itemsForBatching, maxTokens, v.estimator)
if err != nil { if err != nil {
return Result{}, err return Result{}, err
} }
@@ -96,9 +112,10 @@ func (v *LLMBackedValidator) Validate(ctx context.Context, req Request) (Result,
var response LLMValidationResponse var response LLMValidationResponse
responseSchema := responseschema.MustLookup(responseschema.ValidatorDecisionSetKey) responseSchema := responseschema.MustLookup(responseschema.ValidatorDecisionSetKey)
stage := stagename.ValidatorBatch(req.ModuleInstance, v.name, batch.BatchIndex)
call := func(callCtx context.Context) error { call := func(callCtx context.Context) error {
_, err = req.LLMClient.CompleteStructured(callCtx, StructuredCompletionRequest{ _, err = req.LLMClient.CompleteStructured(callCtx, StructuredCompletionRequest{
StageName: fmt.Sprintf("%s:%s:batch-%04d", req.ModuleInstance, v.name, batch.BatchIndex), StageName: stage,
Messages: messages, Messages: messages,
Model: resolvedValidationModel(req.Config, v.model), Model: resolvedValidationModel(req.Config, v.model),
ResponseSchema: &responseSchema, ResponseSchema: &responseSchema,
@@ -112,27 +129,15 @@ func (v *LLMBackedValidator) Validate(ctx context.Context, req Request) (Result,
} }
artifacts := InteractionArtifacts{} artifacts := InteractionArtifacts{}
if req.DiagnosticsWriter != nil { if req.DiagnosticsWriter != nil {
stage := fmt.Sprintf("%s:%s:batch-%04d", req.ModuleInstance, v.name, batch.BatchIndex)
promptMetadata := validatorPromptMetadata(v.validatorType) promptMetadata := validatorPromptMetadata(v.validatorType)
artifacts, _ = req.DiagnosticsWriter.WriteInteraction( artifacts, _ = req.DiagnosticsWriter.WriteInteraction(
stage, stage,
map[string]any{ map[string]any{
"validator_name": v.name, "validator_name": v.name,
"validator_type": v.validatorType, "validator_type": v.validatorType,
"batch_index": batch.BatchIndex, "batch_index": batch.BatchIndex,
"prompt_metadata": map[string]any{ "prompt_metadata": promptMetadata.DiagnosticsMap(),
"prompt_id": promptMetadata.PromptID, "response_schema": responseSchema.DiagnosticsMap(),
"prompt_version": promptMetadata.PromptVersion,
"prompt_source": promptMetadata.PromptSource,
"embedded_path": promptMetadata.EmbeddedPath,
"sha256": promptMetadata.SHA256,
},
"response_schema": map[string]any{
"id": responseSchema.ID,
"version": responseSchema.Version,
"name": responseSchema.Name,
"sha256": responseSchema.SHA256,
},
}, },
map[string]any{"messages": messages, "items": batch.Items}, map[string]any{"messages": messages, "items": batch.Items},
response, response,
@@ -140,12 +145,19 @@ func (v *LLMBackedValidator) Validate(ctx context.Context, req Request) (Result,
) )
} }
if err != nil { if err != nil {
if structuredoutput.IsMalformedError(err) {
llmDecisions = append(llmDecisions, rejectBatch(batch.Items, ReasonValidatorMalformed, fmt.Sprintf("validator response malformed: %s", strings.TrimSpace(err.Error())))...)
warnings = append(warnings, newValidatorWarning(v.name, batch.BatchIndex, ReasonValidatorMalformed, err.Error(), artifacts))
continue
}
return Result{}, fmt.Errorf("LLM validator %q completion failed: %w", v.name, err) return Result{}, fmt.Errorf("LLM validator %q completion failed: %w", v.name, err)
} }
batchDecisions, err := mapLLMResponseToDecisions(batch.Items, response) batchDecisions, err := mapLLMResponseToDecisions(batch.Items, response)
if err != nil { if err != nil {
return Result{}, fmt.Errorf("LLM validator %q response invalid: %w", v.name, err) llmDecisions = append(llmDecisions, rejectBatch(batch.Items, ReasonValidatorMalformed, fmt.Sprintf("validator response malformed: %s", strings.TrimSpace(err.Error())))...)
warnings = append(warnings, newValidatorWarning(v.name, batch.BatchIndex, ReasonValidatorMalformed, err.Error(), artifacts))
continue
} }
for i := range batchDecisions { for i := range batchDecisions {
batchDecisions[i].DiagnosticArtifactPath = artifacts.ResponsePayloadPath batchDecisions[i].DiagnosticArtifactPath = artifacts.ResponsePayloadPath
@@ -154,12 +166,13 @@ func (v *LLMBackedValidator) Validate(ctx context.Context, req Request) (Result,
} }
all := append([]Decision(nil), immediate...) all := append([]Decision(nil), immediate...)
all = append(all, oversized...)
all = append(all, llmDecisions...) all = append(all, llmDecisions...)
if err := EnforceDecisionCardinality(req.CandidateProposal, all); err != nil { if err := EnforceDecisionCardinality(req.CandidateProposal, all); err != nil {
return Result{}, err return Result{}, err
} }
sort.SliceStable(all, func(i, j int) bool { return all[i].ProposalIndex < all[j].ProposalIndex }) sort.SliceStable(all, func(i, j int) bool { return all[i].ProposalIndex < all[j].ProposalIndex })
return Result{ValidatorName: v.name, Decisions: all}, nil return Result{ValidatorName: v.name, Decisions: all, Warnings: warnings}, nil
} }
func validatorPromptMetadata(validatorType LLMValidatorType) prompts.Metadata { func validatorPromptMetadata(validatorType LLMValidatorType) prompts.Metadata {
@@ -301,3 +314,77 @@ func mapLLMResponseToDecisions(items []LLMValidationItem, response LLMValidation
} }
return decisions, nil return decisions, nil
} }
func oversizedValidationDecisions(items []LLMValidationItem, maxPromptTokens int, estimator chunking.TokenEstimator) []Decision {
out := make([]Decision, 0)
for _, item := range items {
singleTokens, err := estimateBatchTokens(estimator, []LLMValidationItem{item})
if err != nil || singleTokens <= maxPromptTokens {
continue
}
out = append(out, rejection(item.CorrectionIndex, ReasonValidatorInputTooLarge, "validation input exceeds max prompt tokens"))
}
return out
}
func oversizedValidationWarnings(validatorName string, maxPromptTokens int, items []LLMValidationItem, estimator chunking.TokenEstimator) []stagewarnings.StageWarning {
out := make([]stagewarnings.StageWarning, 0)
for _, item := range items {
singleTokens, err := estimateBatchTokens(estimator, []LLMValidationItem{item})
if err != nil || singleTokens <= maxPromptTokens {
continue
}
out = append(out, stagewarnings.StageWarning{
Scope: stagewarnings.ScopeValidator,
ValidatorName: validatorName,
ReasonCode: ReasonValidatorInputTooLarge,
Message: fmt.Sprintf("validation input exceeds max prompt tokens for proposal %d", item.CorrectionIndex),
})
}
return out
}
func filterItemsByDecision(items []LLMValidationItem, decisions []Decision) []LLMValidationItem {
if len(decisions) == 0 {
return append([]LLMValidationItem(nil), items...)
}
rejected := make(map[int]struct{}, len(decisions))
for _, decision := range decisions {
rejected[decision.ProposalIndex] = struct{}{}
}
out := make([]LLMValidationItem, 0, len(items))
for _, item := range items {
if _, ok := rejected[item.CorrectionIndex]; ok {
continue
}
out = append(out, item)
}
return out
}
func rejectBatch(items []LLMValidationItem, reasonCode string, message string) []Decision {
out := make([]Decision, 0, len(items))
for _, item := range items {
out = append(out, rejection(item.CorrectionIndex, reasonCode, message))
}
return out
}
func newValidatorWarning(validatorName string, batchIndex int, reasonCode string, message string, artifacts InteractionArtifacts) stagewarnings.StageWarning {
idx := batchIndex
return stagewarnings.StageWarning{
Scope: stagewarnings.ScopeValidator,
ValidatorName: validatorName,
BatchIndex: &idx,
ReasonCode: reasonCode,
Message: strings.TrimSpace(message),
DiagnosticArtifactPath: diagnosticArtifactPath(artifacts),
}
}
func diagnosticArtifactPath(artifacts InteractionArtifacts) string {
if artifacts.ErrorPayloadPath != "" {
return artifacts.ErrorPayloadPath
}
return artifacts.ResponsePayloadPath
}

View File

@@ -248,29 +248,55 @@ func TestLLMBackedValidatorApprovalAndRejection(t *testing.T) {
} }
} }
func TestLLMBackedValidatorMalformedOutputFails(t *testing.T) { func TestLLMBackedValidatorMalformedOutputRejectsBatch(t *testing.T) {
client := &fakeStructuredLLMClient{err: errors.New("malformed structured output")} client := &fakeStructuredLLMClient{err: errors.New("malformed structured output")}
v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model") v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model")
req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")}) req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")})
req.LLMClient = client req.LLMClient = client
_, err := v.Validate(context.Background(), req) res, err := v.Validate(context.Background(), req)
if err == nil || !strings.Contains(err.Error(), "completion failed") { if err != nil {
t.Fatalf("expected malformed output error, got %v", err) t.Fatalf("expected malformed output downgrade, got %v", err)
}
if len(res.Decisions) != 1 || res.Decisions[0].Approved || res.Decisions[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("unexpected decisions: %+v", res.Decisions)
}
if len(res.Warnings) != 1 || res.Warnings[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("expected malformed warning, got %+v", res.Warnings)
} }
} }
func TestLLMBackedValidatorMissingDecisionFails(t *testing.T) { func TestLLMBackedValidatorProviderMalformedEnvelopeRejectsBatch(t *testing.T) {
client := &fakeStructuredLLMClient{err: errors.New("provider response assistant message content is empty")}
v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model")
req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")})
req.LLMClient = client
res, err := v.Validate(context.Background(), req)
if err != nil {
t.Fatalf("expected malformed provider envelope downgrade, got %v", err)
}
if len(res.Decisions) != 1 || res.Decisions[0].Approved || res.Decisions[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("unexpected decisions: %+v", res.Decisions)
}
if len(res.Warnings) != 1 || res.Warnings[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("expected malformed warning, got %+v", res.Warnings)
}
}
func TestLLMBackedValidatorMissingDecisionRejectsBatch(t *testing.T) {
client := &fakeStructuredLLMClient{responses: []LLMValidationResponse{{Validations: []LLMValidationDecision{}}}} client := &fakeStructuredLLMClient{responses: []LLMValidationResponse{{Validations: []LLMValidationDecision{}}}}
v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model") v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model")
req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")}) req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")})
req.LLMClient = client req.LLMClient = client
_, err := v.Validate(context.Background(), req) res, err := v.Validate(context.Background(), req)
if err == nil || !strings.Contains(err.Error(), "response invalid") { if err != nil {
t.Fatalf("expected missing decision error, got %v", err) t.Fatalf("expected missing decision downgrade, got %v", err)
}
if len(res.Decisions) != 1 || res.Decisions[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("unexpected decisions: %+v", res.Decisions)
} }
} }
func TestLLMBackedValidatorDuplicateDecisionFails(t *testing.T) { func TestLLMBackedValidatorDuplicateDecisionRejectsBatch(t *testing.T) {
client := &fakeStructuredLLMClient{responses: []LLMValidationResponse{{Validations: []LLMValidationDecision{ client := &fakeStructuredLLMClient{responses: []LLMValidationResponse{{Validations: []LLMValidationDecision{
{CorrectionIndex: 0, Approved: true, Confidence: 0.9, Reason: "ok"}, {CorrectionIndex: 0, Approved: true, Confidence: 0.9, Reason: "ok"},
{CorrectionIndex: 0, Approved: false, Confidence: 0.9, Reason: "dup"}, {CorrectionIndex: 0, Approved: false, Confidence: 0.9, Reason: "dup"},
@@ -278,20 +304,74 @@ func TestLLMBackedValidatorDuplicateDecisionFails(t *testing.T) {
v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model") v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model")
req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")}) req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")})
req.LLMClient = client req.LLMClient = client
_, err := v.Validate(context.Background(), req) res, err := v.Validate(context.Background(), req)
if err == nil || !strings.Contains(err.Error(), "duplicate") { if err != nil {
t.Fatalf("expected duplicate decision error, got %v", err) t.Fatalf("expected duplicate decision downgrade, got %v", err)
}
if len(res.Decisions) != 1 || res.Decisions[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("unexpected decisions: %+v", res.Decisions)
} }
} }
func TestLLMBackedValidatorUnknownProposalIndexFails(t *testing.T) { func TestLLMBackedValidatorUnknownProposalIndexRejectsBatch(t *testing.T) {
client := &fakeStructuredLLMClient{responses: []LLMValidationResponse{{Validations: []LLMValidationDecision{{CorrectionIndex: 99, Approved: true, Confidence: 0.9, Reason: "unknown"}}}}} client := &fakeStructuredLLMClient{responses: []LLMValidationResponse{{Validations: []LLMValidationDecision{{CorrectionIndex: 99, Approved: true, Confidence: 0.9, Reason: "unknown"}}}}}
v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model") v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model")
req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")}) req := makeReq([]proposals.EnrichedCorrectionProposal{mk(0, "gestures", "Jesters")})
req.LLMClient = client req.LLMClient = client
_, err := v.Validate(context.Background(), req) res, err := v.Validate(context.Background(), req)
if err == nil || !strings.Contains(err.Error(), "unknown") { if err != nil {
t.Fatalf("expected unknown index error, got %v", err) t.Fatalf("expected unknown index downgrade, got %v", err)
}
if len(res.Decisions) != 1 || res.Decisions[0].ReasonCode != ReasonValidatorMalformed {
t.Fatalf("unexpected decisions: %+v", res.Decisions)
}
}
func TestLLMBackedValidatorOversizedSingleProposalRejectsOnlyThatProposal(t *testing.T) {
client := &fakeStructuredLLMClient{responses: []LLMValidationResponse{{Validations: []LLMValidationDecision{
{CorrectionIndex: 1, Approved: true, Confidence: 0.9, Reason: "ok"},
}}}}
v, _ := NewLLMBackedValidator("spoken_form_plausibility_review", LLMValidatorTypeSpokenFormPlausibility, "test-model")
huge := strings.Repeat("gestures ", 200)
req := Request{
WorkingTranscript: &schema.Transcript{Segments: []schema.Segment{
{ID: 1, Text: huge},
{ID: 2, Text: "There were gestures at the temple.", Categories: []string{"narration"}},
}},
CandidateProposal: []proposals.EnrichedCorrectionProposal{
{
CorrectionProposal: proposals.CorrectionProposal{TargetSegmentID: 1, OriginalText: huge, CorrectedText: "Jesters", Confidence: 0.9},
ProposalMetadata: proposals.ProposalMetadata{ProposalIndex: 0, ModuleKey: "homophones", ModuleInstance: "homophones"},
},
{
CorrectionProposal: proposals.CorrectionProposal{TargetSegmentID: 2, OriginalText: "gestures", CorrectedText: "Jesters", Confidence: 0.9},
ProposalMetadata: proposals.ProposalMetadata{ProposalIndex: 1, ModuleKey: "homophones", ModuleInstance: "homophones"},
},
},
ModuleKey: "homophones",
ModuleInstance: "homophones",
ReplacementPolicy: proposals.ReplacementPolicyRequireUnique,
}
req.LLMClient = client
cfg := config.Default()
cfg.ValidationMaxPromptTokens = 200
req.Config = &cfg
res, err := v.Validate(context.Background(), req)
if err != nil {
t.Fatalf("expected oversize downgrade, got %v", err)
}
if len(res.Decisions) != 2 {
t.Fatalf("expected two decisions, got %+v", res.Decisions)
}
if res.Decisions[0].ReasonCode != ReasonValidatorInputTooLarge || res.Decisions[0].Approved {
t.Fatalf("expected first decision oversize rejection, got %+v", res.Decisions[0])
}
if !res.Decisions[1].Approved {
t.Fatalf("expected second decision approved, got %+v", res.Decisions[1])
}
if len(res.Warnings) != 1 || res.Warnings[0].ReasonCode != ReasonValidatorInputTooLarge {
t.Fatalf("expected one oversize warning, got %+v", res.Warnings)
} }
} }

View File

@@ -6,18 +6,25 @@ import (
"strings" "strings"
"gitea.maximumdirect.net/eric/audita/internal/core/config" "gitea.maximumdirect.net/eric/audita/internal/core/config"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
"gitea.maximumdirect.net/eric/audita/internal/core/schema" "gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals" "gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
stagewarnings "gitea.maximumdirect.net/eric/audita/internal/framework/warnings"
) )
const ( const (
ReasonApproved = "approved" ReasonApproved = "approved"
ReasonLowConfidence = "low_confidence" ReasonLowConfidence = "low_confidence"
ReasonMissingOriginalText = "missing_original_text" ReasonMissingOriginalText = "missing_original_text"
ReasonMissingTargetSegment = "missing_target_segment" ReasonMissingTargetSegment = "missing_target_segment"
ReasonEmptyCorrectedText = "empty_corrected_text" ReasonEmptyResultingText = "empty_resulting_segment"
ReasonNoEffect = "no_effect" ReasonNoEffect = "no_effect"
ReasonProtectedGlossaryTerm = "protected_glossary_term" ReasonProtectedGlossaryTerm = "protected_glossary_term"
ReasonInvalidTargetSegment = "invalid_target_segment_id"
ReasonEmptyOriginalText = "empty_original_text"
ReasonInvalidConfidence = "invalid_confidence"
ReasonValidatorMalformed = "validator_response_malformed"
ReasonValidatorInputTooLarge = "validator_input_too_large"
) )
// Request is the runtime input shared by deterministic validators. // Request is the runtime input shared by deterministic validators.
@@ -45,8 +52,9 @@ type Decision struct {
// Result is one validator output containing exactly one decision per proposal index. // Result is one validator output containing exactly one decision per proposal index.
type Result struct { type Result struct {
ValidatorName string `json:"validator_name"` ValidatorName string `json:"validator_name"`
Decisions []Decision `json:"decisions"` Decisions []Decision `json:"decisions"`
Warnings []stagewarnings.StageWarning `json:"warnings,omitempty"`
} }
// ValidationScheduler provides bounded execution for validator LLM calls. // ValidationScheduler provides bounded execution for validator LLM calls.
@@ -111,13 +119,13 @@ func confidenceThresholdForModule(moduleKey string, cfg *config.Config) float64
return 0.0 return 0.0
} }
switch moduleKey { switch moduleKey {
case "glossary": case modulecatalog.KeyGlossary:
return cfg.Thresholds.Glossary return cfg.Thresholds.Glossary
case "grammar": case modulecatalog.KeyGrammar:
return cfg.Thresholds.Grammar return cfg.Thresholds.Grammar
case "homophones": case modulecatalog.KeyHomophones:
return cfg.Thresholds.Homophones return cfg.Thresholds.Homophones
case "spoken_word": case modulecatalog.KeySpokenWord:
return cfg.Thresholds.SpokenWord return cfg.Thresholds.SpokenWord
default: default:
return 0.0 return 0.0

View File

@@ -68,6 +68,31 @@ func TestConfidenceThresholdValidator(t *testing.T) {
} }
} }
func TestProposalShapeValidator(t *testing.T) {
req := Request{CandidateProposal: []proposals.EnrichedCorrectionProposal{
mkCandidate(0, 1, "teh", "the", 0.9),
mkCandidate(1, 0, "teh", "the", 0.9),
mkCandidate(2, 1, " ", "the", 0.9),
mkCandidate(3, 1, "teh", "the", 1.5),
}}
res, err := (ProposalShapeValidator{}).Validate(context.Background(), req)
if err != nil {
t.Fatalf("Validate error: %v", err)
}
if !res.Decisions[0].Approved {
t.Fatalf("expected proposal 0 approved")
}
if res.Decisions[1].ReasonCode != ReasonInvalidTargetSegment {
t.Fatalf("expected invalid target segment rejection, got %+v", res.Decisions[1])
}
if res.Decisions[2].ReasonCode != ReasonEmptyOriginalText {
t.Fatalf("expected empty original rejection, got %+v", res.Decisions[2])
}
if res.Decisions[3].ReasonCode != ReasonInvalidConfidence {
t.Fatalf("expected invalid confidence rejection, got %+v", res.Decisions[3])
}
}
func TestOriginalTextPresenceValidator(t *testing.T) { func TestOriginalTextPresenceValidator(t *testing.T) {
req := Request{WorkingTranscript: &schema.Transcript{Segments: []schema.Segment{{ID: 1, Text: "hello world"}}}, CandidateProposal: []proposals.EnrichedCorrectionProposal{ req := Request{WorkingTranscript: &schema.Transcript{Segments: []schema.Segment{{ID: 1, Text: "hello world"}}}, CandidateProposal: []proposals.EnrichedCorrectionProposal{
mkCandidate(0, 1, "hello", "hi", 0.9), mkCandidate(0, 1, "hello", "hi", 0.9),
@@ -90,10 +115,13 @@ func TestOriginalTextPresenceValidator(t *testing.T) {
} }
func TestNonEmptyCorrectionValidator(t *testing.T) { func TestNonEmptyCorrectionValidator(t *testing.T) {
req := Request{CandidateProposal: []proposals.EnrichedCorrectionProposal{ req := Request{
mkCandidate(0, 1, "hello", "hi", 0.9), WorkingTranscript: &schema.Transcript{Segments: []schema.Segment{{ID: 1, Text: "hello world"}}},
mkCandidate(1, 1, "hello", " ", 0.9), ReplacementPolicy: proposals.ReplacementPolicyRequireUnique,
}} CandidateProposal: []proposals.EnrichedCorrectionProposal{
mkCandidate(0, 1, "hello", "hi", 0.9),
mkCandidate(1, 1, "hello world", " ", 0.9),
}}
res, err := (NonEmptyCorrectionValidator{}).Validate(context.Background(), req) res, err := (NonEmptyCorrectionValidator{}).Validate(context.Background(), req)
if err != nil { if err != nil {
t.Fatalf("Validate error: %v", err) t.Fatalf("Validate error: %v", err)
@@ -101,8 +129,8 @@ func TestNonEmptyCorrectionValidator(t *testing.T) {
if !res.Decisions[0].Approved { if !res.Decisions[0].Approved {
t.Fatalf("expected proposal 0 approved") t.Fatalf("expected proposal 0 approved")
} }
if res.Decisions[1].ReasonCode != ReasonEmptyCorrectedText { if res.Decisions[1].ReasonCode != ReasonEmptyResultingText {
t.Fatalf("expected empty_corrected_text, got %+v", res.Decisions[1]) t.Fatalf("expected empty_resulting_segment, got %+v", res.Decisions[1])
} }
} }
@@ -151,9 +179,14 @@ func TestStableReasonCodes(t *testing.T) {
ReasonLowConfidence, ReasonLowConfidence,
ReasonMissingOriginalText, ReasonMissingOriginalText,
ReasonMissingTargetSegment, ReasonMissingTargetSegment,
ReasonEmptyCorrectedText, ReasonEmptyResultingText,
ReasonNoEffect, ReasonNoEffect,
ReasonProtectedGlossaryTerm, ReasonProtectedGlossaryTerm,
ReasonInvalidTargetSegment,
ReasonEmptyOriginalText,
ReasonInvalidConfidence,
ReasonValidatorMalformed,
ReasonValidatorInputTooLarge,
} }
for _, code := range codes { for _, code := range codes {
if strings.TrimSpace(code) == "" { if strings.TrimSpace(code) == "" {

View File

@@ -0,0 +1,18 @@
package warnings
type Scope string
const (
ScopeProposalGeneration Scope = "proposal_generation"
ScopeValidator Scope = "validator"
)
type StageWarning struct {
Scope Scope `json:"scope"`
ReasonCode string `json:"reason_code"`
Message string `json:"message"`
SectionIndex *int `json:"section_index,omitempty"`
ValidatorName string `json:"validator_name,omitempty"`
BatchIndex *int `json:"batch_index,omitempty"`
DiagnosticArtifactPath string `json:"diagnostic_artifact_path,omitempty"`
}

View File

@@ -2,12 +2,11 @@ package glossary
import ( import (
"context" "context"
"fmt"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts" "gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation" "gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals" "gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/prompts"
builtinvalidators "gitea.maximumdirect.net/eric/audita/internal/validators" builtinvalidators "gitea.maximumdirect.net/eric/audita/internal/validators"
) )
@@ -36,65 +35,10 @@ func (m *Module) Validators() []contracts.Validator {
return append([]contracts.Validator(nil), m.validators...) return append([]contracts.Validator(nil), m.validators...)
} }
func (m *Module) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { func (m *Module) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
sectionTranscript := transcriptForSection(req.WorkingTranscript, req.Section) return proposal_generation.ExecuteModuleProposal(ctx, proposal_generation.ModuleProposalRequest{
sectionIndex := 0 ProposalRequest: req,
if req.Section != nil { PromptID: prompts.PromptIDModuleGlossaryProposal,
sectionIndex = req.Section.Index BuildMessages: BuildProposalMessages,
}
transcriptDescription := ""
if req.Config != nil {
transcriptDescription = req.Config.TranscriptDescription
}
messages, err := BuildProposalMessages(sectionTranscript, req.Glossary, sectionIndex, transcriptDescription)
if err != nil {
return nil, err
}
generated, err := proposal_generation.GenerateCandidates(ctx, proposal_generation.Request{
ModuleKey: req.RunSpec.ModuleKey,
ModuleInstance: req.RunSpec.InstanceName,
ReplacementPolicy: req.RunSpec.ReplacementPolicy,
WorkingTranscript: req.WorkingTranscript,
Section: req.Section,
Glossary: req.Glossary,
Config: req.Config,
Messages: messages,
PromptMetadata: map[string]any{
"prompt_id": proposalPromptMetadata().PromptID,
"prompt_version": proposalPromptMetadata().PromptVersion,
"prompt_source": proposalPromptMetadata().PromptSource,
"embedded_path": proposalPromptMetadata().EmbeddedPath,
"sha256": proposalPromptMetadata().SHA256,
},
StageName: proposalStageName(req),
StartIndex: 0,
LLMClient: req.LLMClient,
Scheduler: req.LLMScheduler,
DiagnosticsDir: req.DiagnosticsDir,
}) })
if err != nil {
return nil, err
}
return generated.Corrections, nil
}
func transcriptForSection(transcript *schema.Transcript, section *contracts.SectionMetadata) *schema.Transcript {
if section == nil || transcript == nil {
return transcript
}
segments := make([]schema.Segment, 0, len(transcript.Segments))
for _, seg := range transcript.Segments {
if seg.ID >= section.StartSegmentID && seg.ID <= section.EndSegmentID {
segments = append(segments, seg)
}
}
return &schema.Transcript{Segments: segments}
}
func proposalStageName(req contracts.ProposalRequest) string {
if req.Section == nil || req.Section.Index == 0 {
return fmt.Sprintf("%s:proposal", req.RunSpec.InstanceName)
}
return fmt.Sprintf("%s:proposal:section-%04d", req.RunSpec.InstanceName, req.Section.Index)
} }

View File

@@ -129,6 +129,7 @@ func TestGlossaryModuleValidatorChain(t *testing.T) {
got = append(got, v.Name()) got = append(got, v.Name())
} }
want := []string{ want := []string{
"proposal_shape",
"no_effect", "no_effect",
"original_text_presence", "original_text_presence",
"confidence_threshold", "confidence_threshold",
@@ -185,7 +186,7 @@ func TestGlossaryModuleProposeMapsCorrectionsAndWritesDiagnostics(t *testing.T)
if len(client.calls) != 1 || client.calls[0].StageName != "glossary:proposal" { if len(client.calls) != 1 || client.calls[0].StageName != "glossary:proposal" {
t.Fatalf("expected one glossary:proposal call, got %+v", client.calls) t.Fatalf("expected one glossary:proposal call, got %+v", client.calls)
} }
if len(out) != 1 || out[0].CorrectedText != "Jesters" { if len(out.Proposals) != 1 || out.Proposals[0].CorrectedText != "Jesters" {
t.Fatalf("unexpected proposals: %+v", out) t.Fatalf("unexpected proposals: %+v", out)
} }
diagFiles, globErr := filepath.Glob(filepath.Join(diagDir, "glossary", "*proposal*response-payload.json")) diagFiles, globErr := filepath.Glob(filepath.Join(diagDir, "glossary", "*proposal*response-payload.json"))

View File

@@ -10,43 +10,13 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/prompts" "gitea.maximumdirect.net/eric/audita/internal/prompts"
) )
type promptSegment struct {
ID int `json:"id"`
Speaker string `json:"speaker"`
Start float64 `json:"start"`
End float64 `json:"end"`
Text string `json:"text"`
Categories []string `json:"categories,omitempty"`
}
type promptTranscriptSection struct {
SectionIndex int `json:"section_index"`
Segments []promptSegment `json:"segments"`
}
func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) { func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
glossaryJSON, err := json.MarshalIndent(glossary, "", " ") glossaryJSON, err := json.MarshalIndent(glossary, "", " ")
if err != nil { if err != nil {
return nil, fmt.Errorf("marshal glossary prompt context: %w", err) return nil, fmt.Errorf("marshal glossary prompt context: %w", err)
} }
sectionPayload := promptTranscriptSection{ sectionJSON, err := promptcontext.MarshalTranscriptSectionJSON(transcript, sectionIndex)
SectionIndex: sectionIndex,
Segments: make([]promptSegment, 0),
}
if transcript != nil {
for _, s := range transcript.Segments {
sectionPayload.Segments = append(sectionPayload.Segments, promptSegment{
ID: s.ID,
Speaker: s.Speaker,
Start: s.Start,
End: s.End,
Text: s.Text,
Categories: append([]string(nil), s.Categories...),
})
}
}
sectionJSON, err := json.MarshalIndent(sectionPayload, "", " ")
if err != nil { if err != nil {
return nil, fmt.Errorf("marshal transcript prompt context: %w", err) return nil, fmt.Errorf("marshal transcript prompt context: %w", err)
} }
@@ -65,7 +35,3 @@ func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Gloss
{Role: "user", Content: user}, {Role: "user", Content: user},
}, nil }, nil
} }
func proposalPromptMetadata() prompts.Metadata {
return prompts.MustLookupMetadata(prompts.PromptIDModuleGlossaryProposal)
}

View File

@@ -2,12 +2,11 @@ package grammar
import ( import (
"context" "context"
"fmt"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts" "gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation" "gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals" "gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/prompts"
builtinvalidators "gitea.maximumdirect.net/eric/audita/internal/validators" builtinvalidators "gitea.maximumdirect.net/eric/audita/internal/validators"
) )
@@ -36,65 +35,10 @@ func (m *Module) Validators() []contracts.Validator {
return append([]contracts.Validator(nil), m.validators...) return append([]contracts.Validator(nil), m.validators...)
} }
func (m *Module) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { func (m *Module) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
sectionTranscript := transcriptForSection(req.WorkingTranscript, req.Section) return proposal_generation.ExecuteModuleProposal(ctx, proposal_generation.ModuleProposalRequest{
sectionIndex := 0 ProposalRequest: req,
if req.Section != nil { PromptID: prompts.PromptIDModuleGrammarProposal,
sectionIndex = req.Section.Index BuildMessages: BuildProposalMessages,
}
transcriptDescription := ""
if req.Config != nil {
transcriptDescription = req.Config.TranscriptDescription
}
messages, err := BuildProposalMessages(sectionTranscript, req.Glossary, sectionIndex, transcriptDescription)
if err != nil {
return nil, err
}
generated, err := proposal_generation.GenerateCandidates(ctx, proposal_generation.Request{
ModuleKey: req.RunSpec.ModuleKey,
ModuleInstance: req.RunSpec.InstanceName,
ReplacementPolicy: req.RunSpec.ReplacementPolicy,
WorkingTranscript: req.WorkingTranscript,
Section: req.Section,
Glossary: req.Glossary,
Config: req.Config,
Messages: messages,
PromptMetadata: map[string]any{
"prompt_id": proposalPromptMetadata().PromptID,
"prompt_version": proposalPromptMetadata().PromptVersion,
"prompt_source": proposalPromptMetadata().PromptSource,
"embedded_path": proposalPromptMetadata().EmbeddedPath,
"sha256": proposalPromptMetadata().SHA256,
},
StageName: proposalStageName(req),
StartIndex: 0,
LLMClient: req.LLMClient,
Scheduler: req.LLMScheduler,
DiagnosticsDir: req.DiagnosticsDir,
}) })
if err != nil {
return nil, err
}
return generated.Corrections, nil
}
func transcriptForSection(transcript *schema.Transcript, section *contracts.SectionMetadata) *schema.Transcript {
if section == nil || transcript == nil {
return transcript
}
segments := make([]schema.Segment, 0, len(transcript.Segments))
for _, seg := range transcript.Segments {
if seg.ID >= section.StartSegmentID && seg.ID <= section.EndSegmentID {
segments = append(segments, seg)
}
}
return &schema.Transcript{Segments: segments}
}
func proposalStageName(req contracts.ProposalRequest) string {
if req.Section == nil || req.Section.Index == 0 {
return fmt.Sprintf("%s:proposal", req.RunSpec.InstanceName)
}
return fmt.Sprintf("%s:proposal:section-%04d", req.RunSpec.InstanceName, req.Section.Index)
} }

View File

@@ -131,6 +131,7 @@ func TestGrammarModuleValidatorChain(t *testing.T) {
got = append(got, v.Name()) got = append(got, v.Name())
} }
want := []string{ want := []string{
"proposal_shape",
"no_effect", "no_effect",
"original_text_presence", "original_text_presence",
"confidence_threshold", "confidence_threshold",
@@ -187,7 +188,7 @@ func TestGrammarModuleProposeMapsCorrectionsAndWritesDiagnostics(t *testing.T) {
if len(client.calls) != 1 || client.calls[0].StageName != "grammar:proposal" { if len(client.calls) != 1 || client.calls[0].StageName != "grammar:proposal" {
t.Fatalf("expected one grammar:proposal call, got %+v", client.calls) t.Fatalf("expected one grammar:proposal call, got %+v", client.calls)
} }
if len(out) != 1 || out[0].CorrectedText != "Hello, world" { if len(out.Proposals) != 1 || out.Proposals[0].CorrectedText != "Hello, world" {
t.Fatalf("unexpected proposals: %+v", out) t.Fatalf("unexpected proposals: %+v", out)
} }
diagFiles, globErr := filepath.Glob(filepath.Join(diagDir, "grammar", "*proposal*response-payload.json")) diagFiles, globErr := filepath.Glob(filepath.Join(diagDir, "grammar", "*proposal*response-payload.json"))

View File

@@ -10,20 +10,6 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/prompts" "gitea.maximumdirect.net/eric/audita/internal/prompts"
) )
type promptSegment struct {
ID int `json:"id"`
Speaker string `json:"speaker"`
Start float64 `json:"start"`
End float64 `json:"end"`
Text string `json:"text"`
Categories []string `json:"categories,omitempty"`
}
type promptTranscriptSection struct {
SectionIndex int `json:"section_index"`
Segments []promptSegment `json:"segments"`
}
// BuildProposalMessages constrains corrections to punctuation/capitalization/ // BuildProposalMessages constrains corrections to punctuation/capitalization/
// spacing cleanup with strict meaning guards. // spacing cleanup with strict meaning guards.
func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) { func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
@@ -32,24 +18,7 @@ func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Gloss
return nil, fmt.Errorf("marshal glossary prompt context: %w", err) return nil, fmt.Errorf("marshal glossary prompt context: %w", err)
} }
sectionPayload := promptTranscriptSection{ sectionJSON, err := promptcontext.MarshalTranscriptSectionJSON(transcript, sectionIndex)
SectionIndex: sectionIndex,
Segments: make([]promptSegment, 0),
}
if transcript != nil {
for _, s := range transcript.Segments {
sectionPayload.Segments = append(sectionPayload.Segments, promptSegment{
ID: s.ID,
Speaker: s.Speaker,
Start: s.Start,
End: s.End,
Text: s.Text,
Categories: append([]string(nil), s.Categories...),
})
}
}
sectionJSON, err := json.MarshalIndent(sectionPayload, "", " ")
if err != nil { if err != nil {
return nil, fmt.Errorf("marshal transcript prompt context: %w", err) return nil, fmt.Errorf("marshal transcript prompt context: %w", err)
} }
@@ -68,7 +37,3 @@ func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Gloss
{Role: "user", Content: user}, {Role: "user", Content: user},
}, nil }, nil
} }
func proposalPromptMetadata() prompts.Metadata {
return prompts.MustLookupMetadata(prompts.PromptIDModuleGrammarProposal)
}

View File

@@ -2,12 +2,11 @@ package homophones
import ( import (
"context" "context"
"fmt"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts" "gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation" "gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals" "gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/prompts"
builtinvalidators "gitea.maximumdirect.net/eric/audita/internal/validators" builtinvalidators "gitea.maximumdirect.net/eric/audita/internal/validators"
) )
@@ -36,65 +35,10 @@ func (m *Module) Validators() []contracts.Validator {
return append([]contracts.Validator(nil), m.validators...) return append([]contracts.Validator(nil), m.validators...)
} }
func (m *Module) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { func (m *Module) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
sectionTranscript := transcriptForSection(req.WorkingTranscript, req.Section) return proposal_generation.ExecuteModuleProposal(ctx, proposal_generation.ModuleProposalRequest{
sectionIndex := 0 ProposalRequest: req,
if req.Section != nil { PromptID: prompts.PromptIDModuleHomophonesProposal,
sectionIndex = req.Section.Index BuildMessages: BuildProposalMessages,
}
transcriptDescription := ""
if req.Config != nil {
transcriptDescription = req.Config.TranscriptDescription
}
messages, err := BuildProposalMessages(sectionTranscript, req.Glossary, sectionIndex, transcriptDescription)
if err != nil {
return nil, err
}
generated, err := proposal_generation.GenerateCandidates(ctx, proposal_generation.Request{
ModuleKey: req.RunSpec.ModuleKey,
ModuleInstance: req.RunSpec.InstanceName,
ReplacementPolicy: req.RunSpec.ReplacementPolicy,
WorkingTranscript: req.WorkingTranscript,
Section: req.Section,
Glossary: req.Glossary,
Config: req.Config,
Messages: messages,
PromptMetadata: map[string]any{
"prompt_id": proposalPromptMetadata().PromptID,
"prompt_version": proposalPromptMetadata().PromptVersion,
"prompt_source": proposalPromptMetadata().PromptSource,
"embedded_path": proposalPromptMetadata().EmbeddedPath,
"sha256": proposalPromptMetadata().SHA256,
},
StageName: proposalStageName(req),
StartIndex: 0,
LLMClient: req.LLMClient,
Scheduler: req.LLMScheduler,
DiagnosticsDir: req.DiagnosticsDir,
}) })
if err != nil {
return nil, err
}
return generated.Corrections, nil
}
func transcriptForSection(transcript *schema.Transcript, section *contracts.SectionMetadata) *schema.Transcript {
if section == nil || transcript == nil {
return transcript
}
segments := make([]schema.Segment, 0, len(transcript.Segments))
for _, seg := range transcript.Segments {
if seg.ID >= section.StartSegmentID && seg.ID <= section.EndSegmentID {
segments = append(segments, seg)
}
}
return &schema.Transcript{Segments: segments}
}
func proposalStageName(req contracts.ProposalRequest) string {
if req.Section == nil || req.Section.Index == 0 {
return fmt.Sprintf("%s:proposal", req.RunSpec.InstanceName)
}
return fmt.Sprintf("%s:proposal:section-%04d", req.RunSpec.InstanceName, req.Section.Index)
} }

View File

@@ -145,6 +145,7 @@ func TestHomophonesModuleValidatorChain(t *testing.T) {
got = append(got, v.Name()) got = append(got, v.Name())
} }
want := []string{ want := []string{
"proposal_shape",
"no_effect", "no_effect",
"original_text_presence", "original_text_presence",
"confidence_threshold", "confidence_threshold",
@@ -201,7 +202,7 @@ func TestHomophonesModuleProposeMapsCorrectionsAndWritesDiagnostics(t *testing.T
if len(client.calls) != 1 || client.calls[0].StageName != "homophones:proposal" { if len(client.calls) != 1 || client.calls[0].StageName != "homophones:proposal" {
t.Fatalf("expected one homophones:proposal call, got %+v", client.calls) t.Fatalf("expected one homophones:proposal call, got %+v", client.calls)
} }
if len(out) != 1 || out[0].CorrectedText != "Jesters" { if len(out.Proposals) != 1 || out.Proposals[0].CorrectedText != "Jesters" {
t.Fatalf("unexpected proposals: %+v", out) t.Fatalf("unexpected proposals: %+v", out)
} }
diagFiles, globErr := filepath.Glob(filepath.Join(diagDir, "homophones", "*proposal*response-payload.json")) diagFiles, globErr := filepath.Glob(filepath.Join(diagDir, "homophones", "*proposal*response-payload.json"))

View File

@@ -10,20 +10,6 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/prompts" "gitea.maximumdirect.net/eric/audita/internal/prompts"
) )
type promptSegment struct {
ID int `json:"id"`
Speaker string `json:"speaker"`
Start float64 `json:"start"`
End float64 `json:"end"`
Text string `json:"text"`
Categories []string `json:"categories,omitempty"`
}
type promptTranscriptSection struct {
SectionIndex int `json:"section_index"`
Segments []promptSegment `json:"segments"`
}
// BuildProposalMessages constrains corrections to conservative homophone and // BuildProposalMessages constrains corrections to conservative homophone and
// mistranscription updates. // mistranscription updates.
func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) { func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
@@ -32,23 +18,7 @@ func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Gloss
return nil, fmt.Errorf("marshal glossary prompt context: %w", err) return nil, fmt.Errorf("marshal glossary prompt context: %w", err)
} }
sectionPayload := promptTranscriptSection{ sectionJSON, err := promptcontext.MarshalTranscriptSectionJSON(transcript, sectionIndex)
SectionIndex: sectionIndex,
Segments: make([]promptSegment, 0),
}
if transcript != nil {
for _, s := range transcript.Segments {
sectionPayload.Segments = append(sectionPayload.Segments, promptSegment{
ID: s.ID,
Speaker: s.Speaker,
Start: s.Start,
End: s.End,
Text: s.Text,
Categories: append([]string(nil), s.Categories...),
})
}
}
sectionJSON, err := json.MarshalIndent(sectionPayload, "", " ")
if err != nil { if err != nil {
return nil, fmt.Errorf("marshal transcript prompt context: %w", err) return nil, fmt.Errorf("marshal transcript prompt context: %w", err)
} }
@@ -67,7 +37,3 @@ func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Gloss
{Role: "user", Content: user}, {Role: "user", Content: user},
}, nil }, nil
} }
func proposalPromptMetadata() prompts.Metadata {
return prompts.MustLookupMetadata(prompts.PromptIDModuleHomophonesProposal)
}

View File

@@ -2,12 +2,11 @@ package spoken_word
import ( import (
"context" "context"
"fmt"
"gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts" "gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation" "gitea.maximumdirect.net/eric/audita/internal/framework/proposal_generation"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals" "gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
"gitea.maximumdirect.net/eric/audita/internal/prompts"
builtinvalidators "gitea.maximumdirect.net/eric/audita/internal/validators" builtinvalidators "gitea.maximumdirect.net/eric/audita/internal/validators"
) )
@@ -36,65 +35,10 @@ func (m *Module) Validators() []contracts.Validator {
return append([]contracts.Validator(nil), m.validators...) return append([]contracts.Validator(nil), m.validators...)
} }
func (m *Module) Propose(ctx context.Context, req contracts.ProposalRequest) ([]proposals.CorrectionProposal, error) { func (m *Module) Propose(ctx context.Context, req contracts.ProposalRequest) (contracts.ProposalResult, error) {
sectionTranscript := transcriptForSection(req.WorkingTranscript, req.Section) return proposal_generation.ExecuteModuleProposal(ctx, proposal_generation.ModuleProposalRequest{
sectionIndex := 0 ProposalRequest: req,
if req.Section != nil { PromptID: prompts.PromptIDModuleSpokenWordProposal,
sectionIndex = req.Section.Index BuildMessages: BuildProposalMessages,
}
transcriptDescription := ""
if req.Config != nil {
transcriptDescription = req.Config.TranscriptDescription
}
messages, err := BuildProposalMessages(sectionTranscript, req.Glossary, sectionIndex, transcriptDescription)
if err != nil {
return nil, err
}
generated, err := proposal_generation.GenerateCandidates(ctx, proposal_generation.Request{
ModuleKey: req.RunSpec.ModuleKey,
ModuleInstance: req.RunSpec.InstanceName,
ReplacementPolicy: req.RunSpec.ReplacementPolicy,
WorkingTranscript: req.WorkingTranscript,
Section: req.Section,
Glossary: req.Glossary,
Config: req.Config,
Messages: messages,
PromptMetadata: map[string]any{
"prompt_id": proposalPromptMetadata().PromptID,
"prompt_version": proposalPromptMetadata().PromptVersion,
"prompt_source": proposalPromptMetadata().PromptSource,
"embedded_path": proposalPromptMetadata().EmbeddedPath,
"sha256": proposalPromptMetadata().SHA256,
},
StageName: proposalStageName(req),
StartIndex: 0,
LLMClient: req.LLMClient,
Scheduler: req.LLMScheduler,
DiagnosticsDir: req.DiagnosticsDir,
}) })
if err != nil {
return nil, err
}
return generated.Corrections, nil
}
func transcriptForSection(transcript *schema.Transcript, section *contracts.SectionMetadata) *schema.Transcript {
if section == nil || transcript == nil {
return transcript
}
segments := make([]schema.Segment, 0, len(transcript.Segments))
for _, seg := range transcript.Segments {
if seg.ID >= section.StartSegmentID && seg.ID <= section.EndSegmentID {
segments = append(segments, seg)
}
}
return &schema.Transcript{Segments: segments}
}
func proposalStageName(req contracts.ProposalRequest) string {
if req.Section == nil || req.Section.Index == 0 {
return fmt.Sprintf("%s:proposal", req.RunSpec.InstanceName)
}
return fmt.Sprintf("%s:proposal:section-%04d", req.RunSpec.InstanceName, req.Section.Index)
} }

View File

@@ -132,6 +132,7 @@ func TestSpokenWordModuleValidatorChain(t *testing.T) {
got = append(got, v.Name()) got = append(got, v.Name())
} }
want := []string{ want := []string{
"proposal_shape",
"no_effect", "no_effect",
"original_text_presence", "original_text_presence",
"confidence_threshold", "confidence_threshold",
@@ -188,7 +189,7 @@ func TestSpokenWordModuleProposeMapsCorrectionsAndWritesDiagnostics(t *testing.T
if len(client.calls) != 1 || client.calls[0].StageName != "spoken_word:proposal" { if len(client.calls) != 1 || client.calls[0].StageName != "spoken_word:proposal" {
t.Fatalf("expected one spoken_word:proposal call, got %+v", client.calls) t.Fatalf("expected one spoken_word:proposal call, got %+v", client.calls)
} }
if len(out) != 1 || out[0].CorrectedText != "I think" { if len(out.Proposals) != 1 || out.Proposals[0].CorrectedText != "I think" {
t.Fatalf("unexpected proposals: %+v", out) t.Fatalf("unexpected proposals: %+v", out)
} }
diagFiles, globErr := filepath.Glob(filepath.Join(diagDir, "spoken_word", "*proposal*response-payload.json")) diagFiles, globErr := filepath.Glob(filepath.Join(diagDir, "spoken_word", "*proposal*response-payload.json"))

View File

@@ -10,20 +10,6 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/prompts" "gitea.maximumdirect.net/eric/audita/internal/prompts"
) )
type promptSegment struct {
ID int `json:"id"`
Speaker string `json:"speaker"`
Start float64 `json:"start"`
End float64 `json:"end"`
Text string `json:"text"`
Categories []string `json:"categories,omitempty"`
}
type promptTranscriptSection struct {
SectionIndex int `json:"section_index"`
Segments []promptSegment `json:"segments"`
}
// BuildProposalMessages constrains corrections to conservative dysfluency // BuildProposalMessages constrains corrections to conservative dysfluency
// cleanup with strict semantic preservation. // cleanup with strict semantic preservation.
func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) { func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Glossary, sectionIndex int, transcriptDescription string) ([]contracts.LLMMessage, error) {
@@ -32,23 +18,7 @@ func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Gloss
return nil, fmt.Errorf("marshal glossary prompt context: %w", err) return nil, fmt.Errorf("marshal glossary prompt context: %w", err)
} }
sectionPayload := promptTranscriptSection{ sectionJSON, err := promptcontext.MarshalTranscriptSectionJSON(transcript, sectionIndex)
SectionIndex: sectionIndex,
Segments: make([]promptSegment, 0),
}
if transcript != nil {
for _, s := range transcript.Segments {
sectionPayload.Segments = append(sectionPayload.Segments, promptSegment{
ID: s.ID,
Speaker: s.Speaker,
Start: s.Start,
End: s.End,
Text: s.Text,
Categories: append([]string(nil), s.Categories...),
})
}
}
sectionJSON, err := json.MarshalIndent(sectionPayload, "", " ")
if err != nil { if err != nil {
return nil, fmt.Errorf("marshal transcript prompt context: %w", err) return nil, fmt.Errorf("marshal transcript prompt context: %w", err)
} }
@@ -67,7 +37,3 @@ func BuildProposalMessages(transcript *schema.Transcript, glossary *schema.Gloss
{Role: "user", Content: user}, {Role: "user", Content: user},
}, nil }, nil
} }
func proposalPromptMetadata() prompts.Metadata {
return prompts.MustLookupMetadata(prompts.PromptIDModuleSpokenWordProposal)
}

View File

@@ -40,6 +40,16 @@ type Metadata struct {
SHA256 string `json:"sha256"` SHA256 string `json:"sha256"`
} }
func (m Metadata) DiagnosticsMap() map[string]any {
return map[string]any{
"prompt_id": m.PromptID,
"prompt_version": m.PromptVersion,
"prompt_source": m.PromptSource,
"embedded_path": m.EmbeddedPath,
"sha256": m.SHA256,
}
}
type definition struct { type definition struct {
id string id string
version string version string

View File

@@ -107,3 +107,16 @@ func TestRenderedPromptsContainHardening(t *testing.T) {
} }
} }
} }
func TestDiagnosticsMapIncludesStablePromptMetadataShapeForAllPrompts(t *testing.T) {
for _, m := range RegisteredMetadata() {
metadataMap := m.DiagnosticsMap()
if metadataMap["prompt_id"] != m.PromptID ||
metadataMap["prompt_version"] != m.PromptVersion ||
metadataMap["prompt_source"] != m.PromptSource ||
metadataMap["embedded_path"] != m.EmbeddedPath ||
metadataMap["sha256"] != m.SHA256 {
t.Fatalf("unexpected diagnostics metadata map for %q: %+v", m.PromptID, metadataMap)
}
}
}

View File

@@ -0,0 +1,57 @@
package testsupport
import (
"os"
"path/filepath"
"strings"
"testing"
)
func ReadFile(t *testing.T, path string) []byte {
t.Helper()
data, err := os.ReadFile(path)
if err != nil {
t.Fatalf("failed to read file %q: %v", path, err)
}
return data
}
func OnlyRunDir(t *testing.T, workDir string) string {
t.Helper()
entries, err := os.ReadDir(workDir)
if err != nil {
t.Fatalf("failed to read work dir %q: %v", workDir, err)
}
dirs := make([]string, 0, len(entries))
for _, e := range entries {
if e.IsDir() {
dirs = append(dirs, filepath.Join(workDir, e.Name()))
}
}
if len(dirs) != 1 {
t.Fatalf("expected exactly one run dir in %q, found %d", workDir, len(dirs))
}
return dirs[0]
}
func AssertNoSecretInFile(t *testing.T, path, secret string) {
t.Helper()
raw := string(ReadFile(t, path))
if strings.Contains(raw, secret) {
t.Fatalf("secret leaked in %s", path)
}
}
func AssertNoSecretInTree(t *testing.T, root, secret string) {
t.Helper()
_ = filepath.WalkDir(root, func(path string, d os.DirEntry, err error) error {
if err != nil || d == nil || d.IsDir() {
return nil
}
raw, readErr := os.ReadFile(path)
if readErr == nil && strings.Contains(string(raw), secret) {
t.Fatalf("secret leaked in %s", path)
}
return nil
})
}

View File

@@ -2,13 +2,16 @@ package validators
import ( import (
"fmt" "fmt"
"strings"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts" "gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/validators/protected_terms" "gitea.maximumdirect.net/eric/audita/internal/validators/protected_terms"
) )
var builtInChains = map[string][]string{ var builtInChains = map[string][]string{
"glossary": { modulecatalog.KeyGlossary: {
KeyProposalShape,
KeyNoEffect, KeyNoEffect,
KeyOriginalTextPresence, KeyOriginalTextPresence,
KeyConfidenceThreshold, KeyConfidenceThreshold,
@@ -17,7 +20,8 @@ var builtInChains = map[string][]string{
KeySpokenFormPlausibility, KeySpokenFormPlausibility,
KeyMeaningReversalReview, KeyMeaningReversalReview,
}, },
"homophones": { modulecatalog.KeyHomophones: {
KeyProposalShape,
KeyNoEffect, KeyNoEffect,
KeyOriginalTextPresence, KeyOriginalTextPresence,
KeyConfidenceThreshold, KeyConfidenceThreshold,
@@ -26,7 +30,8 @@ var builtInChains = map[string][]string{
KeySpokenFormPlausibility, KeySpokenFormPlausibility,
KeyMeaningReversalReview, KeyMeaningReversalReview,
}, },
"spoken_word": { modulecatalog.KeySpokenWord: {
KeyProposalShape,
KeyNoEffect, KeyNoEffect,
KeyOriginalTextPresence, KeyOriginalTextPresence,
KeyConfidenceThreshold, KeyConfidenceThreshold,
@@ -35,7 +40,8 @@ var builtInChains = map[string][]string{
KeyEditorialReview, KeyEditorialReview,
KeyMeaningReversalReview, KeyMeaningReversalReview,
}, },
"grammar": { modulecatalog.KeyGrammar: {
KeyProposalShape,
KeyNoEffect, KeyNoEffect,
KeyOriginalTextPresence, KeyOriginalTextPresence,
KeyConfidenceThreshold, KeyConfidenceThreshold,
@@ -47,9 +53,10 @@ var builtInChains = map[string][]string{
} }
func BuiltInChainKeys(moduleKey string) ([]string, error) { func BuiltInChainKeys(moduleKey string) ([]string, error) {
keys, ok := builtInChains[moduleKey] key := strings.TrimSpace(moduleKey)
keys, ok := builtInChains[key]
if !ok { if !ok {
return nil, fmt.Errorf("no built-in validator chain for module %q", moduleKey) return nil, fmt.Errorf("no built-in validator chain for module %q", key)
} }
out := make([]string, len(keys)) out := make([]string, len(keys))
copy(out, keys) copy(out, keys)
@@ -57,6 +64,7 @@ func BuiltInChainKeys(moduleKey string) ([]string, error) {
} }
func ResolveBuiltInChain(moduleKey string, registry *Registry) ([]contracts.Validator, error) { func ResolveBuiltInChain(moduleKey string, registry *Registry) ([]contracts.Validator, error) {
moduleKey = strings.TrimSpace(moduleKey)
keys, err := BuiltInChainKeys(moduleKey) keys, err := BuiltInChainKeys(moduleKey)
if err != nil { if err != nil {
return nil, err return nil, err
@@ -67,7 +75,7 @@ func ResolveBuiltInChain(moduleKey string, registry *Registry) ([]contracts.Vali
out := make([]contracts.Validator, 0, len(keys)) out := make([]contracts.Validator, 0, len(keys))
for _, key := range keys { for _, key := range keys {
if moduleKey == "glossary" && key == KeyProtectedTerms { if moduleKey == modulecatalog.KeyGlossary && key == KeyProtectedTerms {
// Glossary stages preserve current stricter protection semantics while // Glossary stages preserve current stricter protection semantics while
// reporting the stable protected_terms key. // reporting the stable protected_terms key.
v, buildErr := protected_terms.NewGlossaryStage() v, buildErr := protected_terms.NewGlossaryStage()

View File

@@ -1,6 +1,10 @@
package metadata package metadata
import "gitea.maximumdirect.net/eric/audita/internal/framework/contracts" import (
"strings"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
)
type ExecutionClass string type ExecutionClass string
@@ -9,6 +13,31 @@ const (
ExecutionClassLLMBacked ExecutionClass = "llm_backed" ExecutionClassLLMBacked ExecutionClass = "llm_backed"
) )
const (
KeyProposalShape = "proposal_shape"
KeyConfidenceThreshold = "confidence_threshold"
KeyOriginalTextPresence = "original_text_presence"
KeyNonEmptyCorrectedText = "non_empty_corrected_text"
KeyNoEffect = "no_effect"
KeyProtectedTerms = "protected_terms"
KeySpokenFormPlausibility = "spoken_form_plausibility"
KeyMeaningReversalReview = "meaning_reversal_review"
KeyEditorialReview = "editorial_review"
)
var executionClassByKey = map[string]ExecutionClass{
KeyProposalShape: ExecutionClassDeterministic,
KeyConfidenceThreshold: ExecutionClassDeterministic,
KeyOriginalTextPresence: ExecutionClassDeterministic,
KeyNonEmptyCorrectedText: ExecutionClassDeterministic,
KeyNoEffect: ExecutionClassDeterministic,
KeyProtectedTerms: ExecutionClassDeterministic,
KeySpokenFormPlausibility: ExecutionClassLLMBacked,
KeyMeaningReversalReview: ExecutionClassLLMBacked,
KeyEditorialReview: ExecutionClassLLMBacked,
}
type ClassifiedValidator interface { type ClassifiedValidator interface {
contracts.Validator contracts.Validator
ExecutionClass() ExecutionClass ExecutionClass() ExecutionClass
@@ -18,9 +47,11 @@ func ClassOf(v contracts.Validator) ExecutionClass {
if v == nil { if v == nil {
return ExecutionClassDeterministic return ExecutionClassDeterministic
} }
classFromKey := ClassForKey(v.Name())
classified, ok := v.(ClassifiedValidator) classified, ok := v.(ClassifiedValidator)
if !ok { if !ok {
return ExecutionClassDeterministic return classFromKey
} }
switch classified.ExecutionClass() { switch classified.ExecutionClass() {
case ExecutionClassLLMBacked: case ExecutionClassLLMBacked:
@@ -28,8 +59,16 @@ func ClassOf(v contracts.Validator) ExecutionClass {
case ExecutionClassDeterministic: case ExecutionClassDeterministic:
return ExecutionClassDeterministic return ExecutionClassDeterministic
default: default:
return classFromKey
}
}
func ClassForKey(key string) ExecutionClass {
class, ok := executionClassByKey[strings.TrimSpace(key)]
if !ok {
return ExecutionClassDeterministic return ExecutionClassDeterministic
} }
return class
} }
func Wrap(v contracts.Validator, class ExecutionClass) contracts.Validator { func Wrap(v contracts.Validator, class ExecutionClass) contracts.Validator {

View File

@@ -16,6 +16,16 @@ func (u unclassifiedValidator) Validate(_ context.Context, _ contracts.Validatio
return frameworkvalidators.Result{ValidatorName: u.Name(), Decisions: nil}, nil return frameworkvalidators.Result{ValidatorName: u.Name(), Decisions: nil}, nil
} }
type namedUnclassifiedValidator struct {
name string
}
func (n namedUnclassifiedValidator) Name() string { return n.name }
func (n namedUnclassifiedValidator) Validate(_ context.Context, _ contracts.ValidationRequest) (frameworkvalidators.Result, error) {
return frameworkvalidators.Result{ValidatorName: n.Name(), Decisions: nil}, nil
}
func TestClassOfDefaultsToDeterministic(t *testing.T) { func TestClassOfDefaultsToDeterministic(t *testing.T) {
if got := ClassOf(unclassifiedValidator{}); got != ExecutionClassDeterministic { if got := ClassOf(unclassifiedValidator{}); got != ExecutionClassDeterministic {
t.Fatalf("expected deterministic default class, got %q", got) t.Fatalf("expected deterministic default class, got %q", got)
@@ -28,3 +38,22 @@ func TestWrapExposesExecutionClass(t *testing.T) {
t.Fatalf("expected llm_backed class, got %q", got) t.Fatalf("expected llm_backed class, got %q", got)
} }
} }
func TestClassForKey(t *testing.T) {
if got := ClassForKey(KeyProposalShape); got != ExecutionClassDeterministic {
t.Fatalf("expected deterministic class for %q, got %q", KeyProposalShape, got)
}
if got := ClassForKey(KeySpokenFormPlausibility); got != ExecutionClassLLMBacked {
t.Fatalf("expected llm_backed class for %q, got %q", KeySpokenFormPlausibility, got)
}
if got := ClassForKey("unknown"); got != ExecutionClassDeterministic {
t.Fatalf("expected deterministic fallback for unknown key, got %q", got)
}
}
func TestClassOfFallsBackToStableValidatorKey(t *testing.T) {
v := namedUnclassifiedValidator{name: KeyEditorialReview}
if got := ClassOf(v); got != ExecutionClassLLMBacked {
t.Fatalf("expected llm_backed fallback by key, got %q", got)
}
}

View File

@@ -0,0 +1,13 @@
package proposal_shape
import (
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
frameworkvalidators "gitea.maximumdirect.net/eric/audita/internal/framework/validators"
validatormetadata "gitea.maximumdirect.net/eric/audita/internal/validators/metadata"
)
const Key = "proposal_shape"
func New() (contracts.Validator, error) {
return validatormetadata.Wrap(frameworkvalidators.ProposalShapeValidator{}, validatormetadata.ExecutionClassDeterministic), nil
}

View File

@@ -8,29 +8,31 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/validators/confidence_threshold" "gitea.maximumdirect.net/eric/audita/internal/validators/confidence_threshold"
"gitea.maximumdirect.net/eric/audita/internal/validators/editorial_review" "gitea.maximumdirect.net/eric/audita/internal/validators/editorial_review"
"gitea.maximumdirect.net/eric/audita/internal/validators/meaning_reversal_review" "gitea.maximumdirect.net/eric/audita/internal/validators/meaning_reversal_review"
validatormetadata "gitea.maximumdirect.net/eric/audita/internal/validators/metadata"
"gitea.maximumdirect.net/eric/audita/internal/validators/no_effect" "gitea.maximumdirect.net/eric/audita/internal/validators/no_effect"
"gitea.maximumdirect.net/eric/audita/internal/validators/non_empty_corrected_text" "gitea.maximumdirect.net/eric/audita/internal/validators/non_empty_corrected_text"
"gitea.maximumdirect.net/eric/audita/internal/validators/original_text_presence" "gitea.maximumdirect.net/eric/audita/internal/validators/original_text_presence"
"gitea.maximumdirect.net/eric/audita/internal/validators/proposal_shape"
"gitea.maximumdirect.net/eric/audita/internal/validators/protected_terms" "gitea.maximumdirect.net/eric/audita/internal/validators/protected_terms"
"gitea.maximumdirect.net/eric/audita/internal/validators/spoken_form_plausibility" "gitea.maximumdirect.net/eric/audita/internal/validators/spoken_form_plausibility"
) )
const ( const (
KeyConfidenceThreshold = "confidence_threshold" KeyProposalShape = validatormetadata.KeyProposalShape
KeyOriginalTextPresence = "original_text_presence" KeyConfidenceThreshold = validatormetadata.KeyConfidenceThreshold
KeyNonEmptyCorrectedText = "non_empty_corrected_text" KeyOriginalTextPresence = validatormetadata.KeyOriginalTextPresence
KeyNoEffect = "no_effect" KeyNonEmptyCorrectedText = validatormetadata.KeyNonEmptyCorrectedText
KeyProtectedTerms = "protected_terms" KeyNoEffect = validatormetadata.KeyNoEffect
KeyProtectedTerms = validatormetadata.KeyProtectedTerms
KeySpokenFormPlausibility = "spoken_form_plausibility" KeySpokenFormPlausibility = validatormetadata.KeySpokenFormPlausibility
KeyMeaningReversalReview = "meaning_reversal_review" KeyMeaningReversalReview = validatormetadata.KeyMeaningReversalReview
KeyEditorialReview = "editorial_review" KeyEditorialReview = validatormetadata.KeyEditorialReview
) )
type BuiltInValidatorDefinition struct { type BuiltInValidatorDefinition struct {
Key string Key string
Build func() (contracts.Validator, error) Build func() (contracts.Validator, error)
LLMBacked bool
} }
type Registry struct { type Registry struct {
@@ -39,14 +41,15 @@ type Registry struct {
func NewBuiltInRegistry() *Registry { func NewBuiltInRegistry() *Registry {
defs := []BuiltInValidatorDefinition{ defs := []BuiltInValidatorDefinition{
{Key: KeyProposalShape, Build: proposal_shape.New},
{Key: KeyConfidenceThreshold, Build: confidence_threshold.New}, {Key: KeyConfidenceThreshold, Build: confidence_threshold.New},
{Key: KeyOriginalTextPresence, Build: original_text_presence.New}, {Key: KeyOriginalTextPresence, Build: original_text_presence.New},
{Key: KeyNonEmptyCorrectedText, Build: non_empty_corrected_text.New}, {Key: KeyNonEmptyCorrectedText, Build: non_empty_corrected_text.New},
{Key: KeyNoEffect, Build: no_effect.New}, {Key: KeyNoEffect, Build: no_effect.New},
{Key: KeyProtectedTerms, Build: protected_terms.New}, {Key: KeyProtectedTerms, Build: protected_terms.New},
{Key: KeySpokenFormPlausibility, LLMBacked: true, Build: spoken_form_plausibility.New}, {Key: KeySpokenFormPlausibility, Build: spoken_form_plausibility.New},
{Key: KeyMeaningReversalReview, LLMBacked: true, Build: meaning_reversal_review.New}, {Key: KeyMeaningReversalReview, Build: meaning_reversal_review.New},
{Key: KeyEditorialReview, LLMBacked: true, Build: editorial_review.New}, {Key: KeyEditorialReview, Build: editorial_review.New},
} }
m := make(map[string]BuiltInValidatorDefinition, len(defs)) m := make(map[string]BuiltInValidatorDefinition, len(defs))

View File

@@ -4,6 +4,7 @@ import (
"context" "context"
"testing" "testing"
"gitea.maximumdirect.net/eric/audita/internal/core/modulecatalog"
"gitea.maximumdirect.net/eric/audita/internal/core/schema" "gitea.maximumdirect.net/eric/audita/internal/core/schema"
"gitea.maximumdirect.net/eric/audita/internal/framework/contracts" "gitea.maximumdirect.net/eric/audita/internal/framework/contracts"
"gitea.maximumdirect.net/eric/audita/internal/framework/proposals" "gitea.maximumdirect.net/eric/audita/internal/framework/proposals"
@@ -14,6 +15,7 @@ import (
"gitea.maximumdirect.net/eric/audita/internal/validators/no_effect" "gitea.maximumdirect.net/eric/audita/internal/validators/no_effect"
"gitea.maximumdirect.net/eric/audita/internal/validators/non_empty_corrected_text" "gitea.maximumdirect.net/eric/audita/internal/validators/non_empty_corrected_text"
"gitea.maximumdirect.net/eric/audita/internal/validators/original_text_presence" "gitea.maximumdirect.net/eric/audita/internal/validators/original_text_presence"
"gitea.maximumdirect.net/eric/audita/internal/validators/proposal_shape"
"gitea.maximumdirect.net/eric/audita/internal/validators/protected_terms" "gitea.maximumdirect.net/eric/audita/internal/validators/protected_terms"
"gitea.maximumdirect.net/eric/audita/internal/validators/spoken_form_plausibility" "gitea.maximumdirect.net/eric/audita/internal/validators/spoken_form_plausibility"
) )
@@ -21,6 +23,7 @@ import (
func TestBuiltInRegistryRegistersAllKeys(t *testing.T) { func TestBuiltInRegistryRegistersAllKeys(t *testing.T) {
r := NewBuiltInRegistry() r := NewBuiltInRegistry()
for _, key := range []string{ for _, key := range []string{
KeyProposalShape,
KeyConfidenceThreshold, KeyConfidenceThreshold,
KeyOriginalTextPresence, KeyOriginalTextPresence,
KeyNonEmptyCorrectedText, KeyNonEmptyCorrectedText,
@@ -52,6 +55,7 @@ func TestBuiltInValidatorPackagesConstruct(t *testing.T) {
wantClass validatormetadata.ExecutionClass wantClass validatormetadata.ExecutionClass
} }
cases := []validatorCtor{ cases := []validatorCtor{
{name: "proposal_shape", key: KeyProposalShape, build: proposal_shape.New, wantClass: validatormetadata.ExecutionClassDeterministic},
{name: "confidence_threshold", key: KeyConfidenceThreshold, build: confidence_threshold.New, wantClass: validatormetadata.ExecutionClassDeterministic}, {name: "confidence_threshold", key: KeyConfidenceThreshold, build: confidence_threshold.New, wantClass: validatormetadata.ExecutionClassDeterministic},
{name: "original_text_presence", key: KeyOriginalTextPresence, build: original_text_presence.New, wantClass: validatormetadata.ExecutionClassDeterministic}, {name: "original_text_presence", key: KeyOriginalTextPresence, build: original_text_presence.New, wantClass: validatormetadata.ExecutionClassDeterministic},
{name: "non_empty_corrected_text", key: KeyNonEmptyCorrectedText, build: non_empty_corrected_text.New, wantClass: validatormetadata.ExecutionClassDeterministic}, {name: "non_empty_corrected_text", key: KeyNonEmptyCorrectedText, build: non_empty_corrected_text.New, wantClass: validatormetadata.ExecutionClassDeterministic},
@@ -78,11 +82,6 @@ func TestBuiltInValidatorPackagesConstruct(t *testing.T) {
func TestRegistryBuildsClassifiedValidators(t *testing.T) { func TestRegistryBuildsClassifiedValidators(t *testing.T) {
r := NewBuiltInRegistry() r := NewBuiltInRegistry()
llmKeys := map[string]bool{
KeySpokenFormPlausibility: true,
KeyMeaningReversalReview: true,
KeyEditorialReview: true,
}
for _, key := range r.RegisteredKeys() { for _, key := range r.RegisteredKeys() {
v, err := r.MustBuild(key) v, err := r.MustBuild(key)
if err != nil { if err != nil {
@@ -91,16 +90,28 @@ func TestRegistryBuildsClassifiedValidators(t *testing.T) {
if _, ok := v.(validatormetadata.ClassifiedValidator); !ok { if _, ok := v.(validatormetadata.ClassifiedValidator); !ok {
t.Fatalf("expected built validator %q to expose execution classification metadata", key) t.Fatalf("expected built validator %q to expose execution classification metadata", key)
} }
want := validatormetadata.ExecutionClassDeterministic want := validatormetadata.ClassForKey(key)
if llmKeys[key] {
want = validatormetadata.ExecutionClassLLMBacked
}
if got := validatormetadata.ClassOf(v); got != want { if got := validatormetadata.ClassOf(v); got != want {
t.Fatalf("expected class %q for %q, got %q", want, key, got) t.Fatalf("expected class %q for %q, got %q", want, key, got)
} }
} }
} }
func TestExecutionClassResolvableByStableKeyAndByValidatorInstance(t *testing.T) {
r := NewBuiltInRegistry()
for _, key := range r.RegisteredKeys() {
v, err := r.MustBuild(key)
if err != nil {
t.Fatalf("must build %q: %v", key, err)
}
fromKey := validatormetadata.ClassForKey(key)
fromInstance := validatormetadata.ClassOf(v)
if fromInstance != fromKey {
t.Fatalf("class mismatch for %q: key=%q instance=%q", key, fromKey, fromInstance)
}
}
}
func TestRegistryProtectedTermsUsesNonGlossaryStageBehavior(t *testing.T) { func TestRegistryProtectedTermsUsesNonGlossaryStageBehavior(t *testing.T) {
r := NewBuiltInRegistry() r := NewBuiltInRegistry()
v, err := r.MustBuild(KeyProtectedTerms) v, err := r.MustBuild(KeyProtectedTerms)
@@ -197,7 +208,7 @@ func TestBuiltInRegistryUnknownKeyFails(t *testing.T) {
} }
func TestBuiltInChainKeysResolveForProductionModules(t *testing.T) { func TestBuiltInChainKeysResolveForProductionModules(t *testing.T) {
for _, moduleKey := range []string{"glossary", "homophones", "spoken_word", "grammar"} { for _, moduleKey := range modulecatalog.SupportedKeys() {
keys, err := BuiltInChainKeys(moduleKey) keys, err := BuiltInChainKeys(moduleKey)
if err != nil { if err != nil {
t.Fatalf("resolve keys for %q: %v", moduleKey, err) t.Fatalf("resolve keys for %q: %v", moduleKey, err)
@@ -210,7 +221,7 @@ func TestBuiltInChainKeysResolveForProductionModules(t *testing.T) {
func TestResolveBuiltInChainUsesRegisteredKeys(t *testing.T) { func TestResolveBuiltInChainUsesRegisteredKeys(t *testing.T) {
r := NewBuiltInRegistry() r := NewBuiltInRegistry()
for _, moduleKey := range []string{"glossary", "homophones", "spoken_word", "grammar"} { for _, moduleKey := range modulecatalog.SupportedKeys() {
chain, err := ResolveBuiltInChain(moduleKey, r) chain, err := ResolveBuiltInChain(moduleKey, r)
if err != nil { if err != nil {
t.Fatalf("resolve chain for %q: %v", moduleKey, err) t.Fatalf("resolve chain for %q: %v", moduleKey, err)