130 Commits
v0.8.0 ... main

Author SHA1 Message Date
40f8a1c628 Identify v0.12.0 as the first published application-only release 2026-07-28 15:59:23 -05:00
27d7ad5057 Retire the completed migration roadmaps 2026-07-28 20:04:24 +00:00
1f08a1a94c Record completion of the Promptkit migration 2026-07-28 20:03:11 +00:00
2f42bdde39 Publish Promptkit migration guidance and release notes
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-07-28 19:43:57 +00:00
9a00f30c7b Establish the Scriptorium release procedure 2026-07-28 19:38:11 +00:00
f71d2bbb73 Add an implementation plan and roadmap for Step 9 of the migration plan 2026-07-28 14:25:56 -05:00
6f64947e42 Record the completed Promptkit application cutover 2026-07-28 14:28:26 +00:00
7bb4cf35b9 Document Scriptorium as a Promptkit application 2026-07-28 14:21:19 +00:00
fb0b21c51d Enforce the application dependency boundary 2026-07-28 14:12:36 +00:00
5a00ca81a2 Remove the duplicated prompt framework 2026-07-28 14:07:59 +00:00
e13610481d Adopt Promptkit at application boundaries 2026-07-28 14:04:29 +00:00
309fe9b7ea Add an implementation plan and roadmap for Step 7 of the migration plan 2026-07-28 08:58:35 -05:00
adfd08ffe2 Record Promptkit extraction and publication 2026-07-28 05:01:36 +00:00
c7263ab2a8 Add an implementation plan and roadmap for Step 6 of the migration plan 2026-07-27 23:06:23 -05:00
532c31c09a Complete step 5 of the migration plan and clean up the implemented roadmap 2026-07-27 22:51:11 -05:00
68cd90c657 Complete Promptkit repository foundation 2026-07-28 02:25:20 +00:00
416438d80d Record Promptkit validation and release decision 2026-07-28 02:05:57 +00:00
9298d8ae73 Add an implementation plan and roadmap for Step 5 of the migration plan 2026-07-27 21:01:37 -05:00
4c7278febc Remove the completed step 4 migration roadmap 2026-07-27 20:40:23 -05:00
096208532e Record public facade boundary validation 2026-07-28 01:28:56 +00:00
50bd19b9d1 Strengthen public framework dependency guard 2026-07-28 01:24:31 +00:00
2fc7204bd5 Make HTTP artifact MIME test deterministic 2026-07-28 01:22:31 +00:00
c0d4ea0d4e Preserve public error categories from collaborators 2026-07-28 01:20:41 +00:00
4d7e1327ad Revise the implementation plan to address remaining step 4 gaps 2026-07-27 20:18:02 -05:00
3c33b52b15 Complete the public facade adapter boundary 2026-07-28 00:57:53 +00:00
280916bf4a Move HTTP serving to the public engine 2026-07-28 00:52:24 +00:00
033bc93d3c Use the public engine for CLI run and render 2026-07-28 00:45:59 +00:00
45c2644b9d Add HTTP restricted artifact reader 2026-07-28 00:40:25 +00:00
a74c03bd9b Add public artifact reader support 2026-07-28 00:35:54 +00:00
ad115a2259 Add an implementation plan and roadmap for Step 4 of the migration plan 2026-07-27 19:30:36 -05:00
3074b3165a Complete step 3 of the migration plan and clean up the implemented roadmap 2026-07-27 19:05:28 -05:00
8703793b0c Complete framework characterization baseline 2026-07-27 22:13:14 +00:00
4cb4943a57 Characterize public engine execution behavior 2026-07-27 22:10:54 +00:00
c6c747e94d Move framework tests to contract fixtures 2026-07-27 22:04:08 +00:00
2bbf13e739 Add framework contract test corpus 2026-07-27 21:59:53 +00:00
5edb24a9c1 Add an implementation plan and roadmap for Step 3 of the migration plan 2026-07-27 16:57:08 -05:00
99d5e96316 Accept the Promptkit split architecture 2026-07-26 18:46:42 -05:00
144d840fbe Remove completed documentation roadmap 2026-07-26 18:33:52 -05:00
ed0c9f6370 Implement layered timeout enforcement 2026-07-26 18:31:06 -05:00
a0e905ce46 Complete documentation compliance follow-up 2026-07-26 17:46:01 +00:00
719243e90c Reduce HTTP example test coupling 2026-07-26 17:44:51 +00:00
a9e1b7435c Correct source option documentation 2026-07-26 17:43:42 +00:00
eb6dfb19b0 Clarify OpenAI client timeout precedence 2026-07-26 17:42:44 +00:00
d86b65adad Add documentation refresh follow-up roadmap 2026-07-26 12:38:27 -05:00
31faaf4259 Record documentation refresh completion 2026-07-26 14:27:39 +00:00
9932153b97 Normalize maintained documentation examples 2026-07-26 14:25:44 +00:00
f0ca233c25 Consolidate operational recovery guidance 2026-07-26 14:22:29 +00:00
ff31f8daf8 Refocus internal component documentation 2026-07-26 14:20:06 +00:00
c927b7819d Consolidate external documentation contracts 2026-07-26 14:16:09 +00:00
6d1fb66dd7 Establish canonical developer documentation structure 2026-07-26 14:06:55 +00:00
e0b1d6a0dc Update documentation and testing policies and add a migration plan to cleanly separate the scriptorium CLI from the promptkit internals 2026-07-26 08:59:46 -05:00
33698903be Add deepseek-4-flash profile
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-07-19 08:38:40 -05:00
90b76ddad3 Update copyright statement
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-07-04 22:21:52 -05:00
d5b3d1e061 Remove completed documentation roadmaps 2026-07-05 03:16:11 +00:00
41083de46a Align internal documentation with architecture 2026-07-05 03:13:50 +00:00
07ac7e54c5 Expand consumer integration documentation 2026-07-05 03:09:32 +00:00
879cb021b2 Clarify HTTP operations documentation 2026-07-05 03:06:39 +00:00
574f88bd6a Refresh primary documentation references 2026-07-05 03:02:11 +00:00
d5d7a222a4 Establish canonical documentation links 2026-07-05 02:56:57 +00:00
aabd89aea7 Add a documentation update roadmap 2026-07-04 21:34:03 -05:00
9189cbfc22 Align serve usage flags 2026-07-05 00:24:17 +00:00
872c166ed7 Clarify artifact root symlink behavior 2026-07-05 00:22:54 +00:00
6742def4d3 Redact provider error bodies 2026-07-05 00:20:26 +00:00
1b39f82117 Defer profile extra params validation 2026-07-05 00:19:09 +00:00
f7d821067f Add HTTP size limits 2026-07-05 00:17:07 +00:00
a16f66cbc7 Enforce fs source containment 2026-07-05 00:10:47 +00:00
39485d87f6 Add an implementation plan to reflect follow-up findings from the audit 2026-07-04 19:04:33 -05:00
61e5b0fe58 Document cleanup verification details 2026-07-04 23:41:38 +00:00
f3c21c7d9f Remove stale domain run metadata 2026-07-04 23:39:40 +00:00
bc5f5d3731 Share validator mode handling 2026-07-04 23:38:45 +00:00
93a76f1d36 Share YAML catalog helpers 2026-07-04 23:37:07 +00:00
5c882f26a9 Restrict HTTP file artifact inputs 2026-07-04 23:34:44 +00:00
0d45ac6e3c Audit the internal package API and add a staged improvement roadmap 2026-07-04 18:26:25 -05:00
f7ad756fc3 Document public extra params validation 2026-07-04 23:22:13 +00:00
4fe11b1b2b Clarify public profile and source docs 2026-07-04 23:21:02 +00:00
296f9b1817 Make public options opaque 2026-07-04 23:19:17 +00:00
7a8516b0c6 Redact direct API keys in request formatting 2026-07-04 23:17:44 +00:00
2df2f530b3 Validate public JSON-like inputs 2026-07-04 23:15:50 +00:00
8b25ca72e5 Stop mutating supplied HTTP clients 2026-07-04 23:11:26 +00:00
fa02791fe9 Split prompt and profile load errors 2026-07-04 23:09:23 +00:00
32767b4eb4 Audit the public package API and add a staged improvement roadmap 2026-07-04 18:05:20 -05:00
e1e5351c5d Clean up completed roadmap docs 2026-07-04 17:24:02 -05:00
d60ef66f53 Implement a library profile API and built-in profile docs 2026-07-04 17:23:34 -05:00
4669b73d38 Update OpenAI-compatible auth docs 2026-07-04 17:04:25 +00:00
6f91603168 Add public asset source options 2026-07-04 17:02:09 +00:00
3ad247039b Add request API key support 2026-07-04 16:55:01 +00:00
32e2433628 Add built-in profile repository wiring 2026-07-04 16:48:18 +00:00
712c6b92b8 Add profile repository foundations 2026-07-04 16:41:28 +00:00
89cafcefec Add a roadmap, implementation plan, and built-in profile defaults for a production-ready public library package 2026-07-04 11:38:14 -05:00
1d7fac0a47 Implement fixes to the initial library facade 2026-07-04 11:32:47 -05:00
03d4f27d2b Document public library usage 2026-07-04 14:24:06 +00:00
4ac2038331 Add public run API with injectable LLM 2026-07-04 14:21:37 +00:00
14a7e7e04c Add public prepare engine API 2026-07-04 14:16:19 +00:00
5e522bad8b Add a roadmap and implementation plan for an initial public library package 2026-07-04 09:09:29 -05:00
23872dd742 Implement runtime parameter completion fixes
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-07-04 09:00:19 -05:00
7ffbf5f6ca Update woodpecker config to prepare only linux binaries 2026-07-04 08:53:30 -05:00
d0dc30fcc9 Document runtime provider parameters 2026-07-04 13:29:38 +00:00
b38f7b4dc3 Serialize runtime extra parameters outbound 2026-07-04 13:26:53 +00:00
0512995931 Allow JSON-compatible extra params 2026-07-04 13:24:27 +00:00
049a5feadb Make request execution overrides presence-aware 2026-07-04 13:20:39 +00:00
1798e9c575 Add a roadmap and implementation plan to support reasoning_effort and extra_params in outbound requests 2026-07-04 08:13:10 -05:00
5d4bc8c2b9 Remove completed feature roadmap docs 2026-07-02 20:12:10 -05:00
63fb8fc132 Implement support for OpenRouter sticky routing via a session_id variable 2026-07-02 20:08:44 -05:00
4d4bb7a121 Document prompt cache control behavior 2026-07-02 23:10:24 +00:00
5dcb3cd4fc Expose cache usage in adapters 2026-07-02 23:07:57 +00:00
efe346893c Serialize cache-controlled chat messages 2026-07-02 23:05:56 +00:00
c95d6fcfec Preserve cache control in rendered prompts 2026-07-02 23:03:50 +00:00
0badb4364d Add prompt cache control loading 2026-07-02 23:00:43 +00:00
1f63f8afbb Add a feature roadmap and implementation plan for cache_control values 2026-07-02 17:55:56 -05:00
bc099a31ad Finish cleanup roadmap follow-through
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-26 11:05:08 -05:00
4ff55221a3 Update internal docs for stable runner error reasons and serialized model fields 2026-05-26 15:00:06 +00:00
8d8024099f Complete final cleanup verification and align unsupported artifact test naming 2026-05-26 13:23:45 +00:00
18792fd8d1 Remove unsupported S3 artifact reference placeholder 2026-05-26 13:21:52 +00:00
8860aa033c Reduce CLI test fixture duplication with local setup helpers 2026-05-26 13:20:07 +00:00
3ca14d8b6e Add regression test for JSON schema load failure during prepare 2026-05-26 13:16:26 +00:00
099e9c4a3e Use stable usecase sentinels for HTTP invalid-request mapping 2026-05-26 13:14:57 +00:00
cfe6b9408a Refactor CLI command wiring with shared settings and runner helpers 2026-05-26 13:12:34 +00:00
6ececc749f Centralize YAML catalog scanning for prompt and profile repositories 2026-05-26 13:10:09 +00:00
79901fbb86 Refine execution target mapping helpers and coverage across usecase, HTTP, and LLM 2026-05-26 13:07:35 +00:00
75fa0a030a Added a roadmap to address the issues identifed by the audit 2026-05-26 08:01:33 -05:00
ef64966897 Audit code quality and deduplication opportunities 2026-05-26 07:52:48 -05:00
c6c5e3cb69 Added support for the service_tier key in profiles 2026-05-26 07:44:49 -05:00
3f4fd230b9 Implemented support for loading configuration from nested subdirectories 2026-05-26 07:36:10 -05:00
2091b58066 Completed the documentation update and removed the completed roadmap 2026-05-26 07:15:36 -05:00
c3fe88c9fa Add maintained examples and clean up legacy documentation paths 2026-05-26 03:33:25 +00:00
5830fda516 Add HTTP and OpenAI integration documentation and rewrite Narratio contract 2026-05-26 03:31:01 +00:00
359e910572 Add development policy and internal architecture docs 2026-05-26 03:28:28 +00:00
4950a6bb14 Add operations and troubleshooting documentation 2026-05-26 03:25:34 +00:00
b69ba96811 Rewrite README and add canonical CLI/config documentation 2026-05-26 03:21:11 +00:00
941e2656e8 Add policy documentation and prepare a roadmap to update the remaining documentation accordingly 2026-05-25 22:16:33 -05:00
106 changed files with 4956 additions and 6713 deletions

4
.gitignore vendored
View File

@@ -1,6 +1,5 @@
# ---> Codex # ---> Codex
.codex .codex
AGENTS.md
# ---> Go # ---> Go
# If you prefer the allow list template instead of the deny list, see community template: # If you prefer the allow list template instead of the deny list, see community template:
@@ -57,6 +56,8 @@ mono_crash.*
[Dd]ebugPublic/ [Dd]ebugPublic/
[Rr]elease/ [Rr]elease/
[Rr]eleases/ [Rr]eleases/
!docs/releases/
!docs/releases/*.md
x64/ x64/
x86/ x86/
[Ww][Ii][Nn]32/ [Ww][Ii][Nn]32/
@@ -434,4 +435,3 @@ FodyWeavers.xsd
# JetBrains Rider # JetBrains Rider
*.sln.iml *.sln.iml

View File

@@ -11,9 +11,16 @@ steps:
version="$CI_COMMIT_TAG" version="$CI_COMMIT_TAG"
dist="dist" dist="dist"
pkg="gitea.maximumdirect.net/eric/scriptorium/cmd/scriptorium" pkg="gitea.maximumdirect.net/eric/scriptorium/cmd/scriptorium"
notes="docs/releases/$version.md"
if [ ! -f "$notes" ]; then
printf 'release notes not found: %s\n' "$notes" >&2
exit 1
fi
rm -rf "$dist" rm -rf "$dist"
mkdir -p "$dist" mkdir -p "$dist"
cp "$notes" "$dist/RELEASE_NOTES.md"
build_binary() { build_binary() {
goos="$1" goos="$1"
@@ -22,16 +29,12 @@ steps:
output="$dist/scriptorium-$version-$goos-$goarch$suffix" output="$dist/scriptorium-$version-$goos-$goarch$suffix"
CGO_ENABLED=0 GOOS="$goos" GOARCH="$goarch" \ CGO_ENABLED=0 GOOS="$goos" GOARCH="$goarch" \
go build -trimpath -ldflags "-s -w -X gitea.maximumdirect.net/eric/scriptorium/internal/buildinfo.Version=$version" \ go build -trimpath -ldflags "-s -w" \
-o "$output" "$pkg" -o "$output" "$pkg"
} }
build_binary linux amd64 "" build_binary linux amd64 ""
build_binary linux arm64 "" build_binary linux arm64 ""
build_binary darwin amd64 ""
build_binary darwin arm64 ""
build_binary windows amd64 ".exe"
build_binary windows arm64 ".exe"
- name: publish-release - name: publish-release
image: woodpeckerci/plugin-release image: woodpeckerci/plugin-release
@@ -42,6 +45,7 @@ steps:
from_secret: GITEA_RELEASE_TOKEN from_secret: GITEA_RELEASE_TOKEN
files: files:
- dist/scriptorium-* - dist/scriptorium-*
note: dist/RELEASE_NOTES.md
checksum: sha256 checksum: sha256
checksum-file: SHA256SUMS checksum-file: SHA256SUMS
checksum-flatten: true checksum-flatten: true

1
AGENTS.md Normal file
View File

@@ -0,0 +1 @@
Please review `docs/development.md` for initial orientation in this repository and follow its task-specific reading guide.

View File

@@ -1,4 +1,4 @@
Copyright (c) 2026 eric. Copyright (c) 2026 Eric Rakestraw.
Redistribution and use in source and binary forms, with or without modification, are permitted provided that the following conditions are met: Redistribution and use in source and binary forms, with or without modification, are permitted provided that the following conditions are met:

466
README.md
View File

@@ -1,449 +1,51 @@
# scriptorium # Scriptorium
Scriptorium is a generic prompt execution engine. Scriptorium is a prompt-execution application with a command-line interface and
an HTTP service. It prepares prompt requests, runs them against
OpenAI-compatible model endpoints, and returns generated output with validation
metadata.
It takes: The application uses
- a prompt definition [Promptkit v0.1.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/)
- a selected or default execution profile for prompt, profile, schema, preparation, generation, and validation behavior.
- named input artifacts Scriptorium owns executable configuration, CLI and HTTP mapping, process
- template variables behavior, output presentation, and HTTP artifact-containment policy.
- optional runtime overrides
It returns: ## Quickstart
- for `run`: generated artifact, validation result, metadata
- for `render`: prepared/rendered prompt data (no model output)
## Prompt vs Profile From the repository root:
Scriptorium separates **what** to do (Prompt) from **how** to do it (Profile).
### Prompt Definition
Defines the task logic and output contract.
- Task description and version.
- Message templates (system, user, etc.).
- Required and optional input artifacts.
- Output format and validation rules.
- Repair settings for structured output.
- Optional `default_profile` for convenience.
### Execution Profile
Defines the runtime environment and model settings.
- LLM endpoint (URL).
- Model name.
- Generation parameters: `temperature`, `max_tokens`, `top_p`.
- Runtime settings: `timeout`, `reasoning_effort`.
- API key source via `api_key_env`.
Callers can explicitly provide a `profile_id` to override the prompt's `default_profile`.
## Precedence
Scriptorium uses two precedence layers:
### Application Configuration Precedence
For application-level adapter settings (for example prompt/profile/schema directories, server address, and render output default), precedence is:
1. **CLI Flags**
2. **`config.yml`**
3. **Built-in application defaults**
Application config loading behavior:
- Default config path: `/etc/scriptorium/config.yml`
- Override path: `--config <PATH>` (supported by `run`, `render`, and `serve`)
- If `--config` is provided, the file must exist and be valid.
- If `--config` is omitted, missing `/etc/scriptorium/config.yml` is allowed.
### Runtime Model Precedence
When resolving runtime model settings, Scriptorium follows this precedence model (highest to lowest):
1. **Runtime Overrides**: Provided via CLI flags or HTTP request `model` object.
2. **Execution Profile**: Settings defined in the selected profile.
3. **Application Defaults**: Built-in fallback values.
### Profile Selection Logic
The engine determines which profile to use in this order:
1. Explicit `profile_id` (via `--profile` or HTTP request).
2. The `default_profile` named in the Prompt Definition.
3. Error: If neither is provided and no default exists.
## API Key Policy
To ensure security, Scriptorium does not support raw API keys in configuration files, CLI arguments, or HTTP requests.
- **`api_key_env`**: Profiles and overrides specify the name of an environment variable (e.g., `SCRIPTORIUM_API_KEY`).
- **Runtime Resolution**: The value of the environment variable is read directly from the process environment at runtime.
- **Zero Leakage**: API key values are never included in metadata, logs, or response bodies.
## CLI Usage
Available CLI commands:
- `scriptorium run`
- `scriptorium render`
- `scriptorium serve`
All commands accept `--config <PATH>`.
`prompt_dir` and `profile_dir` may be supplied by CLI flags or `config.yml`:
- `--prompt-dir` or `config.yml` `prompt_dir`
- `--profile-dir` or `config.yml` `profile_dir`
`schema_dir` and `serve` `addr` may also be supplied by `config.yml` where applicable:
- `--schema-dir` or `config.yml` `schema_dir`
- `--addr` or `config.yml` `server.addr`
### `scriptorium run`
Runs a single prompt execution.
**Required Flags:**
- `--prompt`: The prompt ID to execute.
- `--input`: Input mapping `name=path` (repeatable).
**Required Effective Settings:**
- Prompt directory: `--prompt-dir` or `config.yml` `prompt_dir`
- Profile directory: `--profile-dir` or `config.yml` `profile_dir`
**Optional Flags:**
- `--config`: Application config file path. Default discovery path is `/etc/scriptorium/config.yml`.
- `--prompt-dir`: Override prompt directory from config.
- `--profile-dir`: Override profile directory from config.
- `--profile`: Override the prompt's default profile.
- `--var`: Template variable `name=value` (repeatable).
- `--out`: Write output to a file instead of stdout.
- `--llm-base-url`: Override endpoint.
- `--model`: Override model name.
- `--api-key-env`: Override API key environment variable name.
- `--temperature`: Override temperature.
- `--max-tokens`: Override max tokens.
- `--top-p`: Override top_p.
- `--timeout`: Override request timeout (e.g., `30s`, `1m`).
- `--schema-dir`: Base directory for validation schemas.
**Examples:**
Using `config.yml` for prompt/profile directories:
```bash ```bash
scriptorium run \ go run ./cmd/scriptorium render \
--prompt generic.markdown_summary \
--input transcript=./examples/fixtures/transcript.md
```
Overriding config directories explicitly:
```bash
scriptorium run \
--prompt-dir ./prompts \
--profile-dir ./profiles \
--prompt generic.markdown_summary \
--input transcript=./examples/fixtures/transcript.md
```
Overriding the profile:
```bash
scriptorium run \
--prompt-dir ./prompts \
--profile-dir ./profiles \
--prompt generic.markdown_summary \
--profile local-quality \
--input transcript=./examples/fixtures/transcript.md
```
Overriding model and runtime values:
```bash
scriptorium run \
--prompt-dir ./prompts \
--profile-dir ./profiles \
--prompt generic.markdown_summary \
--model gpt-4o \
--temperature 0.7 \
--input transcript=./examples/fixtures/transcript.md
```
Using a local OpenAI-compatible vLLM endpoint:
```bash
scriptorium run \
--prompt-dir ./prompts \
--profile-dir ./profiles \
--prompt generic.markdown_summary \
--llm-base-url http://localhost:8000/v1 \
--model meta-llama-3-8b \
--input transcript=./examples/fixtures/transcript.md
```
### `scriptorium render`
Prepares and renders a prompt without calling the LLM.
`render` uses the same prompt/profile/input/variable/runtime override resolution as `run`:
- Profile selection precedence: `--profile` -> prompt `default_profile` -> error.
- Runtime precedence: CLI runtime overrides -> selected profile -> built-in defaults.
`render` is useful for debugging:
- prompt template rendering
- input mappings
- selected profile behavior
- runtime override behavior
`render` does not:
- call the LLM
- validate model output
- perform repair
- expose resolved API key values
It may include `api_key_env` names where relevant.
**Required Flags:**
- `--prompt`: The prompt ID to render.
- `--input`: Input mapping `name=path` (repeatable).
**Required Effective Settings:**
- Prompt directory: `--prompt-dir` or `config.yml` `prompt_dir`
- Profile directory: `--profile-dir` or `config.yml` `profile_dir`
**Optional Flags:**
- `--config`: Application config file path. Default discovery path is `/etc/scriptorium/config.yml`.
- `--prompt-dir`: Override prompt directory from config.
- `--profile-dir`: Override profile directory from config.
- `--profile`: Override the prompt's default profile.
- `--var`: Template variable `name=value` (repeatable).
- `--out`: Write output to a file instead of stdout.
- `--format`: Render output format (`text` or `json`). Default: `text`.
- `--llm-base-url`: Runtime override for endpoint.
- `--model`: Runtime override for model name.
- `--api-key-env`: Runtime override for API key environment variable name.
- `--temperature`: Runtime override for temperature.
- `--max-tokens`: Runtime override for max tokens.
- `--top-p`: Runtime override for top_p.
- `--timeout`: Runtime override for timeout (e.g., `30s`, `1m`).
**Render Output Formats:**
- `text`: Human-readable output (default).
- `json`: Machine-readable structured output.
Render formatting is modular; additional output formats can be added later without changing prepare/run core logic.
**Examples:**
Default text output using `config.yml` directories:
```bash
scriptorium render \
--prompt generic.markdown_summary \
--input transcript=./examples/fixtures/transcript.md
```
Explicit config path:
```bash
scriptorium render \
--config ./examples/config.yml \ --config ./examples/config.yml \
--prompt generic.markdown_summary \ --prompt generic.markdown_summary \
--input transcript=./examples/fixtures/transcript.md
```
Explicit directory overrides:
```bash
scriptorium render \
--prompt-dir ./prompts \
--profile-dir ./profiles \
--prompt generic.markdown_summary \
--input transcript=./examples/fixtures/transcript.md
```
Explicit JSON output:
```bash
scriptorium render \
--prompt-dir ./prompts \
--profile-dir ./profiles \
--prompt generic.markdown_summary \
--input transcript=./examples/fixtures/transcript.md \ --input transcript=./examples/fixtures/transcript.md \
--input glossary=./examples/fixtures/glossary.yml \
--format json --format json
``` ```
Using prompt `default_profile` (omit `--profile`): This renders the prepared prompt and effective runtime settings without calling
```bash a model. For complete invocation and output behavior, see the
scriptorium render \ [CLI reference](docs/cli.md).
--prompt-dir ./prompts \
--profile-dir ./profiles \
--prompt generic.markdown_summary \
--input transcript=./examples/fixtures/transcript.md
```
Overriding profile selection: ## Documentation
```bash
scriptorium render \
--prompt-dir ./prompts \
--profile-dir ./profiles \
--prompt generic.markdown_summary \
--profile local-quality \
--input transcript=./examples/fixtures/transcript.md
```
Overriding runtime settings: - [CLI reference](docs/cli.md)
```bash - [Configuration reference](docs/config.md)
scriptorium render \ - [HTTP API reference](docs/api.md)
--prompt-dir ./prompts \ - [Operations guide](docs/operations.md)
--profile-dir ./profiles \ - [Consumer integration overview](docs/consumers/api.md)
--prompt generic.markdown_summary \ - [Migration from the former Go package](docs/consumers/migrating-to-promptkit.md)
--input transcript=./examples/fixtures/transcript.md \ - [Subprocess integration](docs/integrations/subprocess.md)
--llm-base-url http://localhost:8000/v1 \ - [Architecture policy](docs/policy/architecture.md)
--model gpt-4o-mini \ - [Promptkit framework formats](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/formats.md)
--temperature 0.2 \ - [Promptkit Go consumer guide](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/consumers/pkg-promptkit.md)
--max-tokens 800 \
--top-p 1.0 \
--timeout 45s
```
Writing rendered output to a file:
```bash
scriptorium render \
--prompt-dir ./prompts \
--profile-dir ./profiles \
--prompt generic.markdown_summary \
--input transcript=./examples/fixtures/transcript.md \
--format text \
--out ./rendered_prompt.txt
```
### `scriptorium serve`
Starts the HTTP API.
**Required Effective Settings:**
- Prompt directory: `--prompt-dir` or `config.yml` `prompt_dir`
- Profile directory: `--profile-dir` or `config.yml` `profile_dir`
**Optional Flags:**
- `--config`: Application config file path. Default discovery path is `/etc/scriptorium/config.yml`.
- `--addr`: Listen address (default `:8080`).
- `--schema-dir`: Base directory for validation schemas.
**Examples:**
Using `config.yml`:
```bash
scriptorium serve
```
Overriding config for local use:
```bash
scriptorium serve \
--prompt-dir ./prompts \
--profile-dir ./profiles \
--addr :9090
```
## HTTP API
### `POST /v1/runs`
Executes a prompt. No built-in authentication is provided; deploy behind a trusted gateway.
**Request Body:**
```json
{
"prompt_id": "generic.structured_events",
"profile_id": "local-quality",
"include_raw_output": false,
"inputs": {
"transcript": {"type": "file", "uri": "./examples/fixtures/transcript.md"}
},
"vars": {
"session_date": "2026-05-04"
},
"model": {
"endpoint": "http://localhost:8000/v1",
"model": "gpt-4o-mini",
"temperature": 0.0
}
}
```
`profile_id` is optional. If omitted, Scriptorium uses the prompt's `default_profile`. If neither is available, the run fails.
**Response:**
Returns a `200 OK` with the generated artifact, validation results, and metadata including the `prompt_id` and the `selected_profile_id`.
**Validation Failures:**
If the model output fails validation (e.g., invalid JSON), the API returns `200 OK` with `validation.status = "failed"`.
**Raw Output Exposure:**
- `raw_model_output` is omitted by default.
- Set `include_raw_output: true` in the request to include it in the response.
- Raw output is preserved internally in run results regardless of HTTP exposure.
## Prompt Definition Authoring
Prompts are defined in YAML.
### Canonical Shape
```yaml
id: generic.structured_events
version: "1.0.0"
description: "Extracts structured events from a transcript"
default_profile: local-quality
inputs:
- name: transcript
required: true
content_type: text/markdown
description: "The raw session transcript"
- name: glossary
required: false
content_type: application/yaml
description: "Optional glossary terms"
messages:
- role: system
content: "You are a helpful assistant."
- role: user
content_file: messages/extract_events.tmpl
output:
format: json
validation_mode: json_schema
schema_path: structured_events.schema.json
repair_attempts: 2
```
**Key Features:**
- **Inline vs File**: Use `content` for short prompts or `content_file` for larger templates. Exactly one must be set per message.
- **Path Resolution**: `content_file` paths are resolved relative to the prompt YAML file.
- **Inputs**: Mark inputs as `required` to ensure the runner fails early if they are missing.
- **Input Metadata**: `content_type` is currently descriptive metadata and not enforced yet.
- **Validation**: Support `none`, `basic`, `json`, and `json_schema`.
- **Repair**: `repair_attempts` enables bounded retries to fix structured output.
## Execution Profile Authoring
Profiles are defined in YAML.
### Canonical Shape
```yaml
id: local-quality
endpoint: http://localhost:8000/v1
model: gpt-4o
temperature: 0.0
max_tokens: 4096
top_p: 1.0
timeout_seconds: 300
reasoning_effort: high
api_key_env: SCRIPTORIUM_API_KEY
```
**Constraints:**
- **No Raw Keys**: Do not include actual API keys. Only specify the environment variable name in `api_key_env`.
- **Local Profiles**: For local endpoints that don't require auth, `api_key_env` can be omitted.
## Examples ## Examples
- **Prompt Definitions**: `prompts/` - [Minimal configuration](examples/config.yml) and
- **Execution Profiles**: `profiles/` [complete configuration](examples/config.full.yml)
- **Schemas**: `schemas/` - [Prompt definitions](examples/prompts/), [execution profiles](examples/profiles/),
- **Fixtures**: `examples/fixtures/` [schemas](examples/schemas/), and [synthetic input fixtures](examples/fixtures/)
- [Render script](examples/render-markdown-summary.sh)
## Build and Test - [HTTP request](examples/http-run.json)
```bash
go build -o scriptorium ./cmd/scriptorium
go test ./...
```

View File

@@ -1,483 +0,0 @@
# Scriptorium Architecture
## 1. Purpose and Non-Goals
Scriptorium is a prompt-definition execution engine.
It accepts named input artifacts, renders prompt templates, calls an LLM, validates output, optionally performs bounded structured-output repair, and returns an artifact with metadata.
Scriptorium also supports rendering/preparing a prompt without calling an LLM. This allows users to inspect the fully rendered prompt messages and effective runtime settings before executing a run.
Scriptorium is not an orchestrator. It must not own transcription, transcript merge/polish steps, notifications, or cross-step workflow control.
For the motivating D&D workflow:
- Narratio orchestrates.
- WhisperX transcribes.
- Seriatim merges transcripts.
- Audita polishes transcripts.
- Scriptorium generates final artifacts from prepared inputs.
Core Go code remains generic.
## 2. Current Architecture
Scriptorium uses a ports-and-adapters architecture to decouple the core execution logic from external dependencies.
### Package Responsibilities
- `cmd/scriptorium`: Binary entrypoint for CLI and HTTP server.
- `internal/domain`: Core domain contracts, including `PromptDefinition`, `ExecutionProfile`, `PreparedRun`, `RunResult`, and related metadata.
- `internal/usecase`: `Runner` use case logic, including prompt preparation, profile selection, runtime override resolution, full run execution, validation, and bounded repair.
- `internal/config`: Application-level config model/loader for adapter settings (for example prompt/profile/schema directories, server address, and render format default).
- `internal/promptdef`: Repository for loading and validating Prompt Definitions from the filesystem.
- `internal/profile`: Repository for loading Execution Profiles from the filesystem.
- `internal/artifact`: Input artifact resolution (`inline`, `file`).
- `internal/prompt`: Template rendering via Go templates.
- `internal/llm`: Provider-neutral client interface and OpenAI-compatible HTTP adapter.
- `internal/validate`: Output validation implementation (`none/basic/json/json_schema`).
- `internal/adapter/cli`: CLI flag parsing, command dispatch, and output handling.
- `internal/adapter/http`: HTTP request/response mapping.
- `internal/format` or equivalent: Formatting of prepared/rendered prompt output for CLI or other adapters, if formatting grows beyond simple CLI-local helpers.
Exact package names may evolve, but the architectural boundaries should remain stable.
### Application Configuration
`config.yml` is adapter/application setup, not domain logic.
Application config is intended for application-level settings such as:
- `prompt_dir`
- `profile_dir`
- `schema_dir`
- `server.addr`
- `defaults.render_format`
Application config precedence is:
1. CLI flags
2. `config.yml`
3. Built-in application defaults
Runtime model settings are intentionally separate:
- Execution profiles and runtime overrides continue to own endpoint/model/runtime behavior.
- `config.yml` does not replace execution profiles.
The core use case (`Runner.Prepare`/`Runner.Run`) does not need to know whether adapter-level settings came from CLI flags or `config.yml`; it receives resolved dependencies and requests from adapters.
## 3. Core Execution Model
Scriptorium has two closely related execution paths:
1. Prepare/render path.
2. Full run path.
The full run path should reuse the prepare path rather than duplicating its logic.
### 3.1 Prepare / Render Data Flow
The prepare path should be represented in the use case layer, preferably as `Runner.Prepare(ctx, RunRequest)` or an equivalent method.
It executes all pre-LLM work:
1. **Validate Request**: Ensure the request includes a prompt ID and all minimum required fields.
2. **Load Prompt Definition**: Retrieve the `PromptDefinition` by ID from the prompt repository.
3. **Select Profile**: Determine the `profile_id` using this precedence:
- Explicit `profile_id` in `RunRequest`.
- `default_profile` specified in the `PromptDefinition`.
- Error if neither is available.
4. **Load Execution Profile**: Retrieve the `ExecutionProfile` from the profile repository.
5. **Resolve Runtime Overrides**: Merge settings based on precedence, highest to lowest:
- Runtime overrides from CLI flags or HTTP request `model` object.
- Execution Profile settings.
- Built-in application defaults.
6. **Resolve Artifacts**: Load all named input artifacts defined in the request.
7. **Render Prompt**: Apply template variables and input artifacts to the prompt templates.
8. **Compute Metadata**: Compute hashes, selected profile ID, prompt ID/version, effective runtime settings, input hashes, rendered prompt hash, and timing information as appropriate.
9. **Return PreparedRun**: Return a `PreparedRun` containing the rendered messages, effective runtime settings, resolved metadata, and input/prompt hashes.
The prepare path must not call the LLM.
The prepare path must not validate model output, because there is no model output.
The prepare path must not perform structured-output repair, because repair only applies after model output exists.
The prepare path should not resolve or expose raw API key values. It may include the selected `api_key_env` name in effective runtime settings or metadata, but never the environment variable value.
### 3.2 Full Run Data Flow
The `Runner.Run(ctx, RunRequest)` flow should reuse the prepare path:
1. **Prepare**: Call the shared prepare flow to load the prompt, select the profile, resolve artifacts, render prompt messages, and compute pre-run metadata.
2. **Call LLM**: Execute the generation request using the effective runtime settings from the prepared run.
3. **Build Output Artifact**: Convert the model response into the configured output artifact.
4. **Validate Output**:
- Validate the model output against the prompt definition's output contract.
- Validation content failures remain successful run results with `validation.status=failed`.
- Validator runtime/config errors are run errors.
5. **Repair If Configured**:
- If structured validation fails and `repair_attempts > 0`, perform bounded repair attempts.
- Re-validate after each repair attempt.
- Repair loops must remain strictly bounded.
6. **Return RunResult**: Produce a `RunResult` containing the final artifact, validation status, raw model output, usage information, and metadata.
`Runner.Run` should not duplicate profile selection, artifact resolution, or prompt rendering logic that already exists in `Runner.Prepare`.
## 4. Domain Model
Key domain types:
- `PromptDefinition`: Defines the "what" of the task: templates, inputs, output contract, validation settings, repair settings, and optional `default_profile`.
- `ExecutionProfile`: Defines the "how" of execution: endpoint, model, generation parameters, timeout, reasoning effort, and `api_key_env`.
- `RunRequest`: The intent to execute or prepare a prompt, including `prompt_id`, optional `profile_id`, inputs, variables, and optional runtime overrides.
- `PreparedRun`: The result of the prepare/render phase. Contains rendered messages, effective runtime settings, selected profile ID, prompt metadata, input hashes, prompt hash, and other pre-LLM metadata.
- `RunResult`: The result of a full run. Contains the generated `Artifact`, `ValidationResult`, raw model output, token usage, and `RunMetadata`.
- `RunMetadata`: Detailed tracing information, including prompt ID/version, selected profile ID, effective model parameters, usage tokens, hashes, timestamps, validation status, and repair attempts where applicable.
- `RenderedPrompt`: Provider-neutral rendered prompt structure.
- `RenderedMessage`: Provider-neutral rendered message with role and content.
- `ArtifactRef`: A reference to an input artifact, such as `file` or `inline`.
- `Artifact`: Loaded artifact content with name, content type, body, source URI, size, and hash.
- `ValidationResult`: Validation status and details for full runs.
`PreparedRun` should be serializable for JSON output and should also be representable in a human-readable text format.
## 5. Interfaces and Adapters
### Primary Ports
- `promptdef.Repository`: Lookup for prompt definitions.
- `profile.Repository`: Lookup for execution profiles.
- `artifact.Reader`: Loading of artifact content.
- `prompt.Renderer`: Template rendering.
- `llm.Client`: Model generation.
- `validate.Validator`: Output validation.
- `format.PreparedRunFormatter` or equivalent: Optional formatting abstraction for rendered/prepared output.
### Current Adapters
- **Repositories**: Filesystem YAML loaders for both prompts and profiles.
- **Artifact Reader**: Composite reader supporting `file` and `inline`.
- **Prompt Renderer**: Go templates with a custom `input` helper.
- **LLM Client**: OpenAI-compatible `/chat/completions` over HTTP.
- **Validator**: Standard validator supporting `none`, `basic`, `json`, and `json_schema`.
- **CLI Adapter**: Supports `run`, `render`, and `serve`.
- **HTTP Adapter**: Supports full run execution through `POST /v1/runs`.
## 6. Public Contracts
### CLI
Scriptorium should expose at least these commands:
- `scriptorium run`: Executes a prompt by preparing it, calling the LLM, validating output, optionally repairing structured output, and returning an artifact.
- `scriptorium render`: Prepares and renders a prompt without calling the LLM.
- `scriptorium serve`: Starts the HTTP API using infrastructure-only flags.
### `scriptorium run`
`run` uses flags such as:
- `--prompt-dir`
- `--profile-dir`
- `--prompt`
- `--profile`
- `--input`
- `--var`
- `--out`
- runtime overrides such as `--model`, `--llm-base-url`, `--temperature`, `--max-tokens`, `--top-p`, `--timeout`, and `--api-key-env` if supported.
`run` should produce the generated artifact as its primary output.
### `scriptorium render`
`render` prepares and renders a prompt without calling an LLM.
It should use the same prompt/profile/input/variable/runtime override flags as `run` where applicable:
- `--prompt-dir`
- `--profile-dir`
- `--prompt`
- `--profile`
- `--input`
- `--var`
- runtime overrides such as `--model`, `--llm-base-url`, `--temperature`, `--max-tokens`, `--top-p`, `--timeout`, and `--api-key-env` if supported.
- `--format`, with initial support for `text` and `json`.
Default render output format should be `text`.
`render` must not call the LLM.
`render` should show the same rendered messages and effective runtime settings that `run` would use.
`render` should include enough information to debug:
- prompt ID
- prompt version
- selected profile ID
- effective runtime settings
- input hashes
- prompt hash
- rendered messages
`render` must not include resolved API key values.
`render` may include the `api_key_env` name.
### Render Output Formats
Initial render output formats:
- `text`: Human-readable default format.
- `json`: Machine-readable structured representation of the prepared run.
Additional formats, such as `markdown`, may be added later.
Render output formatting should be modular. Adding a new output format should not require changing the prepare/run core logic.
The output formatting layer should consume a `PreparedRun` and produce bytes or text for the adapter. It should not reload prompts, re-resolve artifacts, re-render templates, call the LLM, or perform validation.
### `scriptorium serve`
`serve` starts the HTTP API.
It should use infrastructure-only flags such as:
- `--addr`
- `--prompt-dir`
- `--profile-dir`
- `--schema-dir`
`serve` should not introduce a server-level model/runtime precedence layer unless explicitly documented and intentionally implemented.
### HTTP API
Current HTTP API:
- `POST /v1/runs`: Accepts `RunRequest` JSON and returns `RunResponse` JSON. No built-in auth.
Request may include runtime overrides under `model` and an `include_raw_output` boolean.
`raw_model_output` is exposed only when explicitly requested with `include_raw_output=true`.
A future HTTP prepare/render endpoint may be added, such as `POST /v1/renders` or `POST /v1/runs/prepare`, but the initial render feature may be CLI-only. If added later, it should call the same usecase-level prepare path as `scriptorium render`.
### YAML Shapes
Prompt YAML includes:
- `id`
- `version`
- optional `default_profile`
- `inputs`
- `messages`
- `output`
Inputs support:
- `name`
- `required`
- optional `content_type`
- `description`
Messages require:
- `role`
- exactly one of `content` or `content_file`
`content_file` resolves relative to the prompt YAML location.
Profile YAML includes:
- `id`
- `endpoint`
- `model`
- generation parameters
- timeout settings
- reasoning settings
- `api_key_env`
Prompt content must not appear in profile YAML.
Model/runtime/API-key settings must not appear in prompt YAML, except that prompt YAML may specify `default_profile`.
## 7. Render Feature Design
The render feature is a first-class use case, not a CLI-only shortcut.
### Goals
The render feature should help users:
- inspect fully rendered prompt messages
- debug missing inputs
- verify template variable substitution
- verify selected profile resolution
- verify runtime override precedence
- verify file-backed prompt loading
- inspect input hashes and prompt hashes
- prepare for future token budgeting and prompt-size inspection
### Non-Goals
The render feature should not:
- call an LLM
- validate model output
- repair structured output
- resolve or print API key values
- mutate artifacts
- save outputs to artifact storage unless a future explicit output option is added
- become an orchestration step manager
### Usecase Shape
The preferred usecase shape is:
- `Runner.Prepare(ctx, RunRequest) (*PreparedRun, error)`
- `Runner.Run(ctx, RunRequest) (*RunResult, error)`
`Runner.Run` should call `Runner.Prepare`.
The prepare flow should be the only implementation of:
- prompt loading
- profile selection
- runtime override resolution
- artifact resolution
- prompt rendering
- pre-run metadata/hash calculation
### CLI Shape
The preferred command name is `render`.
The command should support:
- `--format text`
- `--format json`
Default format:
- `text`
Unknown formats should produce a clear error.
Formatting should be centralized through a small formatter registry, strategy, switch, or interface so new formats can be added without modifying usecase logic.
### Text Output Expectations
Text output should be optimized for human inspection.
It should include, at minimum:
- prompt ID and version
- selected profile ID
- model name
- endpoint
- effective generation settings
- input hashes
- rendered prompt hash
- rendered messages grouped by role
Text output should be readable and deterministic enough for tests.
It should not include raw API key values.
### JSON Output Expectations
JSON output should be a structured representation of `PreparedRun` or a DTO derived from it.
It should include, at minimum:
- prompt ID and version
- selected profile ID
- effective runtime settings
- input hashes
- rendered prompt hash
- rendered messages
JSON output should not include raw API key values.
JSON output should remain stable enough to be useful for automation and integration tests.
## 8. Guardrails
- **Separation of Concerns**: Prompt content must not belong in execution profiles; model/API settings must not belong in prompt definitions.
- **Security**: Raw API keys are unsupported in all configuration and transport layers. Only `api_key_env` is used.
- **Secret Handling**: Resolved API key values must never appear in rendered output, metadata, logs, HTTP responses, or CLI output.
- **Path Resolution**: `content_file` paths in prompt definitions resolve relative to the prompt YAML file.
- **Integrity**: No silent prompt truncation or omission of content.
- **Reliability**: Repair loops are strictly bounded by `repair_attempts`.
- **No Orchestration Creep**: Scriptorium prepares and executes a single prompt request. It does not coordinate multi-stage workflows.
- **Render Reuse**: The full run path must reuse the prepare/render path to avoid divergent behavior.
- **Formatter Isolation**: Render output formatters must not perform usecase work. They only format a completed `PreparedRun`.
## 9. Testing Strategy
Tests should protect both the run path and the prepare/render path.
### Prepare / Render Tests
Add tests for:
- preparing a prompt with explicit profile selection
- preparing a prompt using `default_profile`
- failing when no explicit profile and no `default_profile` exist
- runtime overrides beating profile values
- profile values beating application defaults
- file-backed prompt bodies rendering correctly
- required inputs failing when missing
- optional inputs being absent when not referenced
- unknown input references failing
- input hashes being included
- rendered prompt hash being included
- effective runtime settings being included
- `api_key_env` name being included where appropriate
- resolved API key values never appearing in `PreparedRun`
- prepare path not calling the LLM
### CLI Render Tests
Add tests for:
- `scriptorium render` mapping flags into `RunRequest`
- default text output
- explicit `--format text`
- explicit `--format json`
- unknown format failure
- text output includes prompt/profile/messages
- JSON output includes prompt/profile/messages
- rendered output never includes resolved API key values
### Run Reuse Tests
Add tests proving:
- `Runner.Run` reuses prepare behavior
- run and render resolve the same prompt/profile/runtime settings for equivalent inputs
- run still validates output
- run still performs bounded repair where configured
### Existing Tests
Continue testing:
- prompt definition loading
- execution profile loading
- artifact loading
- prompt rendering
- LLM adapter behavior
- validation behavior
- HTTP request/response mapping
## 10. Extension Points
Future work should remain grounded in the current architecture:
- **Artifacts**: Add S3 artifact references via a new `artifact.Reader`.
- **LLM**: Implement additional provider adapters, such as Anthropic or Google.
- **Execution**: Add token budgeting, streaming generation, and batch execution capabilities.
- **Prepare/Render**: Add token estimates, prompt-size summaries, or additional render output formats.
- **Repositories**: Implement database-backed repositories for prompts and profiles.
- **Profiles**: Support more granular profile versioning and environment-specific profiles.
- **HTTP**: Add an HTTP prepare/render endpoint if Narratio or another caller needs it.
Future render formats should plug into the formatter layer and should not require changes to the usecase layer.

View File

@@ -0,0 +1,48 @@
# ADR 0001: Adopt Canonical Documentation Ownership
## Status
Accepted
## Date
2026-07-26
## Context
Scriptorium's documentation grew alongside its CLI, HTTP, public Go, and
integration interfaces. As a result, several documents repeated mutable
contracts such as flags, configuration fields, and status behavior. Those
parallel definitions made it unclear which document to update when behavior
changed and increased the risk of documentation drift.
## Decision
Assign each documentation topic one canonical owner, as defined in
[`docs/policy/documentation.md`](../policy/documentation.md). Non-owning
documents may provide short orientation and links, but do not redefine volatile
contracts. Current behavior is documented outside `docs/roadmap/`; roadmaps own
future work, sequencing, and implementation status.
## Alternatives Considered
- Keep broad reference material in several audience-specific documents. This
would preserve local convenience but leave conflicting contract definitions
likely.
- Consolidate all documentation into one reference. This would reduce duplicate
text but would not serve the distinct needs of users, operators, consumers,
and contributors.
## Rationale
Canonical ownership retains audience-specific guidance while making the source
of truth for each contract discoverable. It also makes documentation changes
reviewable alongside the implementation change that requires them.
## Consequences
- Changes to behavior must update the canonical owner in the same change.
- Cross-cutting documentation links to the owner instead of copying its
details.
- Documentation restructuring followed a dedicated implementation roadmap;
repository history, not this ADR, records its completion.

View File

@@ -0,0 +1,288 @@
# ADR 0002: Split Promptkit From Scriptorium
## Status
Accepted
## Date
2026-07-26
## Context
Scriptorium currently combines two products in one Go module:
- a reusable prompt-execution framework with a public Go facade; and
- a runnable application with CLI and HTTP interfaces.
Downstream Go projects increasingly import the framework directly and do not
use the executable interfaces. Keeping both products in one module couples
framework releases, dependencies, documentation, and public API evolution to
application-specific transport concerns.
Promptkit will become the framework project, and Scriptorium will become a slim
application that consumes it. This ADR records that end-state boundary. It does
not assert that the split has been implemented; until then, the current
repository structure and contracts remain authoritative.
## Decision
### Projects And Module Paths
Create a repository named `promptkit` alongside Scriptorium:
| Project | Repository and Go module path | Root Go package |
| --- | --- | --- |
| Promptkit | `gitea.maximumdirect.net/eric/promptkit` | `promptkit` |
| Scriptorium | `gitea.maximumdirect.net/eric/scriptorium` | No reusable root facade after migration |
Promptkit will expose its supported public API from the module root. Its
implementation packages will remain under `internal/` unless a real consumer
extension point requires a public type or interface.
Scriptorium will import only Promptkit's supported public packages. It will not
import Promptkit implementation packages or reproduce Promptkit orchestration.
### Product Responsibilities
Promptkit owns application-neutral framework behavior:
- the engine and its `Prepare` and `Run` workflow;
- public request, result, profile, option, extension, and error APIs;
- prompt-definition loading and rendering;
- profile loading, overlays, and the embedded built-in profile registry;
- schema loading and output validation;
- provider-neutral model-client boundaries and the OpenAI-compatible client;
- artifact types, artifact-reader injection, and general-purpose inline and
caller-selected file readers;
- execution-setting resolution and framework defaults; and
- framework-level secret redaction and error classification.
Scriptorium owns executable and transport behavior:
- the `scriptorium` process and its `run`, `render`, and `serve` commands;
- CLI parsing, streams, output files, formatting, exit codes, and process
cancellation behavior;
- application-configuration discovery and CLI-over-configuration precedence;
- HTTP routing, strict request decoding, DTO mapping, response encoding,
status codes, and transport limits;
- HTTP artifact-root containment and deployment policy;
- server construction, server defaults, and process logging; and
- executable release artifacts.
The dependency direction is:
```text
Scriptorium CLI and HTTP adapters
|
v
Promptkit public API
|
v
injected sources, readers, and model clients
```
### Current Package Disposition
Implementation may reorganize files during extraction, but each current package
has this target owner:
| Current package or file group | Target owner | Disposition |
| --- | --- | --- |
| Root `scriptorium` facade files and tests | Promptkit | Move and rename the public package to `promptkit`; Scriptorium retains no compatibility facade. |
| `internal/domain`, `internal/usecase` | Promptkit | Move as internal engine implementation. |
| `internal/promptdef`, `internal/prompt` | Promptkit | Move as internal prompt loading and rendering. |
| `internal/profile`, `internal/profile/builtin` | Promptkit | Move with embedded built-in assets and registry tests. |
| `internal/filecatalog` | Promptkit | Move as source-loading support. |
| `internal/validate` | Promptkit | Move as schema and output-validation implementation. |
| `internal/llm` | Promptkit | Move with the OpenAI-compatible integration. |
| `internal/artifact` | Split | Move general inline/file reading to Promptkit; keep rooted, denied, and byte-limited HTTP file reading in Scriptorium behind a Promptkit reader interface. |
| `internal/defaults` | Split | Move framework, execution, output-artifact, content-type, and model-client defaults to Promptkit; keep CLI, HTTP, and server defaults in Scriptorium. |
| `internal/adapter/cli`, `internal/adapter/http` | Scriptorium | Keep and refactor to use Promptkit's public API. |
| `internal/config` | Scriptorium | Keep application settings, discovery, validation, and CLI precedence. |
| `internal/format` | Scriptorium | Keep prepared-run presentation, rewritten against Promptkit public values. |
| `cmd/scriptorium` | Scriptorium | Keep as the process entrypoint. |
Tests move with the behavior they protect. Cross-boundary tests will live with
the consuming side: Promptkit protects framework contracts, while Scriptorium
protects adapter mapping, HTTP containment, and executable behavior.
### Public Boundary
Promptkit's initial facade will preserve the useful shape of the current
Scriptorium Go API where that reduces extraction risk. It will expose only the
capabilities required by Promptkit consumers and by Scriptorium:
- engine construction, preparation, and execution;
- public request, result, profile, and error values;
- prompt, profile, schema, artifact-reader, validator, and model-client source
or injection options that have demonstrated consumers; and
- enough stable error identity for Scriptorium to map CLI and HTTP outcomes.
Promptkit will not export its domain package, runner implementation,
repositories, adapter DTOs, or general internal constructors merely to
simplify the move.
Scriptorium's CLI and HTTP adapters will depend on a small consumer-facing
`Prepare`/`Run` interface where test substitution is needed. That interface
belongs at the consuming boundary rather than forcing adapter concepts into
Promptkit.
### Artifact Reading And HTTP Containment
Promptkit will define the artifact-reader extension point used during
preparation. Its ordinary file reader may read a path deliberately supplied by
an in-process or CLI caller and does not claim to be a deployment sandbox.
Scriptorium will implement the HTTP-specific reader that:
- denies file references when no artifact root is configured;
- applies the configured artifact byte limit;
- enforces Scriptorium's documented lexical root-containment rule; and
- maps reader failures to Scriptorium HTTP error responses.
Scriptorium will inject that reader through Promptkit's public construction
boundary. Promptkit will not know about HTTP roots, status codes, request DTOs,
or deployment policy.
### Configuration And Default Ownership
Configuration ownership follows the behavior configured, not the current file
location:
| Configuration category | Owner |
| --- | --- |
| Application configuration discovery, configuration-file precedence, `prompt_dir`, `profile_dir`, and `schema_dir` | Scriptorium |
| CLI flags and their mapping to application settings or request overrides | Scriptorium |
| `server.*`, render-output settings, HTTP byte limits, and server defaults | Scriptorium |
| Prompt-definition, profile, and output-contract file formats | Promptkit |
| Prompt/profile source selection, overlays, schema behavior, and built-in profiles | Promptkit |
| Execution settings, presence-aware request overrides, and execution defaults | Promptkit |
| Built-in OpenAI-compatible client settings, timeout behavior, and provider wire mapping | Promptkit |
| HTTP request and response fields, including their mapping to framework values | Scriptorium |
Scriptorium will translate its application settings and external request
values into Promptkit construction options and requests. When an omitted
Scriptorium setting means “use the framework default,” Scriptorium will omit
the override rather than copy Promptkit's numeric default.
### Compatibility And Versioning
This migration is intentionally breaking:
- new Go consumers will import `gitea.maximumdirect.net/eric/promptkit`;
- Scriptorium will not provide aliases, forwarding wrappers, or deprecated
compatibility packages for its former Go facade;
- existing consumers may remain pinned to the final framework-bearing
Scriptorium tag until migrated; and
- intermediate migration phases need not preserve source compatibility, but
each merged phase must be internally buildable and tested.
Promptkit's first release will be `v0.1.0`. During the migration, incompatible
Promptkit changes may advance its minor version until a stable `v1` contract is
declared. The first slim Scriptorium release will advance the Scriptorium minor
version beyond the final framework-bearing release. Normal semantic-versioning
rules apply independently to both projects after the migration.
Promptkit must be tagged before Scriptorium or another consumer publishes a
release that depends on it. Release branches must use tagged module
dependencies, not local replacements or unpublished revisions.
### Local Development And Cross-Repository Coordination
For coordinated local work, place both repositories in a temporary Go
workspace or use an uncommitted module replacement. `go.work`,
`go.work.sum`, and local filesystem `replace` directives must not be committed
to release branches.
Cross-repository changes follow this order:
1. land and tag the required Promptkit capability;
2. update Scriptorium and other consumers to that tag;
3. run each repository's own CI and smoke checks; and
4. release consumers only after the Promptkit tag is available.
Migration coordination must confirm out-of-band repository creation, Promptkit
tags, and downstream migrations before dependent work proceeds.
Cross-repository changes are coordinated, not treated as atomic commits.
### Documentation And Maintained Assets
Each repository will maintain its own README, contributor guide, architecture,
documentation, testing, release, and operations material appropriate to that
project. Cross-project documents will link to the canonical owner rather than
copy its contract.
Existing documentation and maintained assets have these target owners:
| Current material | Target owner |
| --- | --- |
| Current README and executable quickstart | Scriptorium; Promptkit creates its own framework orientation |
| Public Go package and Go-consumer guidance | Promptkit |
| Prompt, profile, schema, execution-setting, and framework credential reference | Promptkit |
| OpenAI-compatible integration contract and framework internal documents | Promptkit |
| CLI, HTTP API, subprocess, and Scriptorium operations contracts | Scriptorium |
| Consumer interface overview | Scriptorium, revised to route Go consumers to Promptkit |
| Application-configuration discovery, server settings, and adapter internals | Scriptorium |
| Current internal overview and source documentation | Split into repository-local overviews; Promptkit owns framework sources and Scriptorium owns HTTP containment |
| This ADR and cross-project migration records | Scriptorium |
| `examples/go-library` | Promptkit |
| `examples/config*.yml`, `examples/render-markdown-summary.sh`, and `examples/http-run.json` | Scriptorium |
| Example prompts, profiles, schemas, and synthetic fixtures used by the executable examples | Scriptorium |
| Embedded built-in profile assets | Promptkit |
| Scriptorium release workflow and executable packaging | Scriptorium |
| Repository-level license, ignore rules, agent guidance, and development policies | Each repository maintains its own applicable copy |
Promptkit will create or retain its own minimal framework examples and test
fixtures rather than making either repository's tests depend on the other's
working tree. Scriptorium's framework-format documentation will become a short
version-appropriate link to Promptkit, while its maintained executable examples
remain self-contained.
## Alternatives Considered
- Keep the current combined repository and improve package naming. This avoids
migration work but retains release and ownership coupling between the
framework and executable.
- Add Promptkit as a wrapper around the Scriptorium public package. This gives
consumers a new import path but leaves framework ownership and dependency
direction inverted.
- Extract Promptkit while retaining a Scriptorium compatibility facade. This
reduces immediate consumer changes but creates a second public API surface
and prolongs duplicate maintenance.
- Move all artifact reading into Promptkit. This would place HTTP containment,
byte limits, and deployment policy in the application-neutral framework.
- Keep Promptkit and Scriptorium as separate modules in one repository. This
separates imports but not repository permissions, release workflows,
issue ownership, or independent project evolution.
## Rationale
A separate Promptkit project makes the reusable framework the direct owner of
the API that downstream Go projects already consume. Keeping Scriptorium as a
public-API consumer exercises the same boundary as other consumers and prevents
its adapters from relying on framework internals.
The selected split keeps transport and deployment policy close to the
Scriptorium interfaces that expose it, while allowing Promptkit to remain
useful to in-process consumers with different IO and security requirements.
Explicit package, configuration, documentation, and asset ownership reduces
ambiguity during extraction and after release.
## Consequences
- All Go consumers of the framework must change their import path.
- Promptkit and Scriptorium gain independent issue, release, CI, policy, and
documentation lifecycles.
- Scriptorium becomes a real downstream integration test of Promptkit's public
facade.
- Framework changes that affect Scriptorium require tagged, ordered
cross-repository coordination.
- Some current packages, especially artifact reading and defaults, must be
separated by responsibility rather than moved intact.
- Scriptorium's current configuration and documentation references must be
split between application and framework owners.
- Maintainers must inventory and migrate downstream consumers explicitly; no
compatibility facade will hide incomplete migration.
- Until the split is implemented, the current repository structure and
contracts remain authoritative.

View File

@@ -0,0 +1,67 @@
# ADR 0003: Use Maintainer-Run Validation and Tag-Only Releases for Promptkit
## Status
Accepted
## Date
2026-07-28
## Context
[ADR 0002](0002-split-promptkit-from-scriptorium.md) established Promptkit as
an independent Go library with its own repository, version history, validation,
and release coordination. It anticipated independent hosted CI for Promptkit
alongside Scriptorium's existing executable build and CI policy.
Promptkit is presently a single-maintainer library. It does not produce a
runnable command, so executable packaging and binary-release automation do not
apply. Its validation and release model should be explicit before repository
guidance relies on it.
## Decision
Promptkit will use maintainer-run validation rather than hosted CI at this
stage. From a clean checkout, the maintainer will run the repository-documented
test, vet, build, formatting, documentation-link, and repository-hygiene checks
before changes are accepted and before a release tag is published.
Promptkit releases consist of source commits and semantic Go module tags. The
project does not release runnable binaries or maintain binary-packaging
automation.
Scriptorium's executable build, hosted CI, and binary-release policies are
unaffected. The repository boundary, independent version history, release
ordering, and other migration decisions accepted by ADR 0002 remain in force.
Where ADR 0002 anticipated independent hosted CI for Promptkit, this later ADR
controls Promptkit validation.
## Alternatives Considered
- Add hosted Promptkit CI now. This would provide automated remote enforcement,
but its setup and maintenance are not proportionate to the present
single-maintainer library and do not replace the maintainer's release
responsibility.
- Require local Git hooks. Hooks can provide fast feedback, but they are
machine-local, can be bypassed, and are not a durable substitute for the
documented clean-checkout validation procedure.
## Rationale
A documented maintainer-run procedure provides a clear acceptance and release
gate with little operational overhead for the project's current contribution
pattern. If maintenance load or contributor patterns change, a later ADR may
introduce hosted CI without changing Promptkit's library or tag-based release
model.
## Consequences
- Promptkit repository guidance must define the complete local validation
procedure and the checks required before accepting or tagging a change.
- Release evidence is the maintainer's successful clean-checkout validation,
not a hosted CI result.
- Promptkit releases contain source and semantic Go module tags only.
- A future move to hosted CI requires a later architectural decision.
- Scriptorium continues to validate, build, package, and release its executable
under its own policies.

146
docs/api.md Normal file
View File

@@ -0,0 +1,146 @@
# HTTP API Reference
This is the canonical public HTTP contract for Scriptorium.
## Service And Route
`POST /v1/runs` runs one prompt request and returns generated output,
validation, and metadata. The service has no built-in authentication or
authorization; deploy it behind appropriate network and authentication controls.
The service address and HTTP limits are configured as described in the
[configuration reference](config.md). `serve` invocation is defined in the
[CLI reference](cli.md).
Requests and responses are JSON objects. Requests are decoded as JSON regardless
of their `Content-Type`; successful JSON responses use
`Content-Type: application/json`. There are no query parameters.
## Request Limits
The configured request-body limit includes inline artifact bodies. The artifact
limit applies to HTTP `file` inputs. The response limit applies to the encoded
response, including the artifact body and optional raw output. A limit of zero
disables that limit.
A request body over its limit returns `413 request_too_large`; an oversized
file input returns `413 artifact_too_large`; an oversized encoded response
returns `413 response_too_large`.
## `POST /v1/runs`
### Request Body
The maintained [request example](../examples/http-run.json) is a complete
copyable shape. The smallest valid shape is:
```json
{
"prompt_id": "generic.markdown_summary",
"inputs": {
"transcript": {"type": "inline", "body": "Source text"}
}
}
```
| Field | Required | Meaning |
| --- | --- | --- |
| `prompt_id` | yes | Non-blank prompt ID. |
| `prompt_version` | no | Prompt version filter. |
| `profile_id` | no | Execution-profile ID; otherwise the prompt must set `default_profile`. |
| `inputs` | yes | Non-empty object mapping input names to references. |
| `vars` | no | Object mapping template-variable names to strings. |
| `model` | no | Runtime model-override object. |
| `include_raw_output` | no | Include `raw_model_output` when true. |
An input reference has a required `type` of `file` or `inline`. A `file`
reference requires `uri`; an `inline` reference requires `body`.
HTTP file references require a configured artifact root. Relative paths resolve
within that root. Absolute paths must be lexically within it; traversal outside
it is rejected with `400 artifact_not_allowed`. This lexical check does not
resolve symlinks: the operating system follows symlinks inside the root,
including ones that target outside it. Keep the root narrow and inaccessible to
untrusted writers.
The optional `model` object accepts `endpoint`, `model`, `temperature`,
`max_tokens`, `top_p`, `timeout_seconds`, `service_tier`,
`reasoning_effort`, `api_key_env`, and `extra_params`. Numeric ranges and
framework credential semantics are defined by the
[Promptkit format reference](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/formats.md).
Explicit zero values for the numeric fields are overrides; zero
`timeout_seconds` disables the per-generation deadline only, retaining the
request context and configured transport cap. The timeout layers are defined in
the [Promptkit outbound integration contract](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/integrations/openai-compatible-chat.md#timeout-and-cancellation).
Raw API-key values are not accepted. `api_key` and any other unknown model
field cause `400 invalid_json`.
### Strict JSON
Request decoding rejects malformed JSON, unknown fields at every request level,
and trailing JSON tokens with `400 invalid_json`. A blank `prompt_id` or
empty `inputs` object returns `400 invalid_request`.
### Success Response
A completed run returns `200 OK`, including when generated content fails its
validation contract. The response contains:
- `artifact`: `name`, `content_type`, `body`, `size`, `hash`, and
optional `uri`;
- `validation`: `status`, `mode`, `repair_attempts`, `is_valid`, plus
optional `errors` and `schema_path`;
- `metadata`: run, prompt, rendered-prompt, profile, model, input-hash, usage,
timing, validation, and repair-attempt metadata; and
- optional `raw_model_output` when requested.
`metadata.model_params` has `endpoint`, `model`, `temperature`,
`max_tokens`, `top_p`, and `timeout_seconds`, plus optional
`service_tier`, `reasoning_effort`, `api_key_env`, and `extra_params`.
`metadata.usage` always includes `prompt_tokens`, `completion_tokens`,
`total_tokens`, `cached_tokens`, and `cache_write_tokens`; unavailable
cache usage is reported as zero.
A validation failure has `validation.status: "failed"`, `is_valid: false`,
and any available diagnostic errors, while still returning the artifact and
metadata.
## Error Responses
Errors have this shape:
```json
{"error":{"code":"invalid_request","message":"prompt_id is required"}}
```
Messages are concise and do not expose wrapped internal causes.
| Status | Code | Meaning |
| --- | --- | --- |
| `400` | `invalid_json` | Malformed JSON, unknown field, or trailing JSON. |
| `400` | `invalid_request` | Missing or invalid request data or runtime override. |
| `400` | `profile_required` | No profile ID and no prompt default profile. |
| `400` | `prompt_load_failed` | Prompt definition failed to load. |
| `400` | `profile_load_failed` | Profile failed to load. |
| `400` | `artifact_not_allowed` | HTTP file input is disabled or outside the artifact root. |
| `400` | `artifact_read_failed` | Input artifact is invalid or cannot be read. |
| `400` | `prompt_render_failed` | Prompt template rendering failed. |
| `400` | `api_key_env_missing` | The selected credential environment variable is unset or empty. |
| `404` | `not_found` | Route does not exist. |
| `404` | `prompt_not_found` | Prompt ID or version does not exist. |
| `404` | `profile_not_found` | Profile ID does not exist. |
| `405` | `method_not_allowed` | The route does not accept the method. |
| `413` | `request_too_large` | Encoded request exceeds its limit. |
| `413` | `artifact_too_large` | File input exceeds its limit. |
| `413` | `response_too_large` | Encoded response exceeds its limit. |
| `500` | `validation_runtime_failed` | Schema or validator runtime failure. |
| `500` | `internal_error` | Unclassified server failure. |
| `502` | `llm_failed` | Outbound model request failed. |
## Retry And Idempotency
Scriptorium provides no idempotency keys, pagination, caching headers, or rate
limits. Clients may retry transport failures or `5xx` responses only when
their workflow tolerates another model call: a retry can produce different
output and incur another provider request.

148
docs/cli.md Normal file
View File

@@ -0,0 +1,148 @@
# CLI Reference
This is the canonical contract for invoking Scriptorium. Configuration discovery,
precedence, application source locations, and server settings are defined in the
[configuration reference](config.md). The [HTTP API reference](api.md) owns
service request and response behavior.
## Shortest Useful Command
```bash
go run ./cmd/scriptorium render \
--config ./examples/config.yml \
--prompt generic.markdown_summary \
--input transcript=./examples/fixtures/transcript.md \
--input glossary=./examples/fixtures/glossary.yml
```
`render` prepares a request without calling an LLM.
## Commands
- `scriptorium run`: prepare a prompt, call the configured LLM, and write the
generated artifact.
- `scriptorium render`: prepare a prompt and write prepared-run output.
- `scriptorium serve`: start the HTTP server.
All commands accept `--config <path>` and reject positional arguments. An
effective `prompt_dir` is required for every command. Supply it through the
configuration contract or the command's `--prompt-dir` flag.
## `scriptorium run`
```text
scriptorium run [flags]
```
Required flags:
| Flag | Meaning |
| --- | --- |
| `--prompt <id>` | Prompt ID to execute. |
| `--input name=path` | Input file mapping; repeat or use comma-separated mappings. |
Optional flags:
| Flag | Meaning |
| --- | --- |
| `--config <path>` | Application configuration file. |
| `--prompt-dir <dir>` | Prompt-definition directory override. |
| `--profile-dir <dir>` | Custom profile-directory override. |
| `--schema-dir <dir>` | Schema base-directory override. |
| `--profile <id>` | Execution-profile override. |
| `--var name=value` | Template-variable mapping; repeat or use comma-separated mappings. |
| `--out <path>` | Write generated content to this file instead of stdout. |
| `--llm-base-url <url>` | Runtime endpoint override. |
| `--model <name>` | Runtime model override. |
| `--api-key-env <name>` | Runtime API-key environment-variable name override. |
| `--temperature <float>` | Runtime temperature override. |
| `--max-tokens <int>` | Runtime maximum-token override. |
| `--top-p <float>` | Runtime top-p override. |
| `--timeout <duration>` | Runtime timeout override using Go duration syntax. |
Deprecated aliases: `--prompt-id` for `--prompt`, and `--profile-id` for
`--profile`.
Omitted numeric runtime flags preserve the selected effective value; explicit
zero values override it. `--timeout 0s` disables the per-generation deadline
only; the caller context and configured transport cap remain active. CLI
durations are converted to whole seconds by truncation toward zero, so any
duration whose absolute value is below one second becomes an explicit
zero-second override. The timeout layers are defined in the
[Promptkit outbound integration contract](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/integrations/openai-compatible-chat.md#timeout-and-cancellation).
There is no raw API-key flag. Use `--api-key-env`.
## `scriptorium render`
```text
scriptorium render [flags]
```
`--prompt <id>` and at least one `--input name=path` are required. The
following optional flags are supported: `--config`, `--prompt-dir`,
`--profile-dir`, `--profile`, `--var`, `--out`, `--llm-base-url`,
`--model`, `--api-key-env`, `--temperature`, `--max-tokens`, `--top-p`,
`--timeout`, and `--format text|json`. Their meanings match the corresponding
`run` flags; `--format` selects prepared-run output and otherwise uses
`defaults.render_format`.
The same deprecated aliases and numeric/timeout behavior as `run` apply.
`render` does not accept `--schema-dir`; configure `schema_dir` through the
configuration file. It resolves profiles and schemas as part of preparation but
does not call an LLM.
## `scriptorium serve`
```text
scriptorium serve [flags]
```
Optional flags:
| Flag | Meaning |
| --- | --- |
| `--config <path>` | Application configuration file. |
| `--addr <listen-address>` | HTTP listen-address override. |
| `--prompt-dir <dir>` | Prompt-definition directory override. |
| `--profile-dir <dir>` | Custom profile-directory override. |
| `--schema-dir <dir>` | Schema base-directory override. |
| `--artifact-root <dir>` | Root for HTTP `file` input references. |
| `--max-request-bytes <n>` | Maximum encoded HTTP request-body bytes; `0` disables the limit. |
| `--max-artifact-bytes <n>` | Maximum HTTP file-input artifact bytes; `0` disables the limit. |
| `--max-response-bytes <n>` | Maximum encoded HTTP response bytes; `0` disables the limit. |
`serve` accepts no runtime model override flags. HTTP request fields, response
schemas, and error codes are defined in the [HTTP API reference](api.md).
## Input And Variable Syntax
`--input name=path` maps an input name to a local file; `--var name=value`
maps a template variable to a string. Both flags can be repeated or contain
comma-separated mappings. Values may contain `=` after the first separator.
Empty names and values are rejected.
CLI inputs are file references. HTTP inline inputs are defined by the
[HTTP API reference](api.md).
## Output And Exit Behavior
- `run` writes generated content to stdout, or to `--out` when supplied, and
writes a concise summary to stderr.
- `render` writes prepared-run output to stdout, or to `--out` when supplied,
without a success summary.
- `serve` writes startup and server errors to stderr.
Exit statuses:
| Status | Meaning |
| --- | --- |
| `0` | Success. |
| `1` | Parse, configuration, loading, rendering, generation, output-write, or other runtime error. |
| `2` | `run` generated and wrote output, but validation failed. |
## Workflows And Examples
The [maintained render script](../examples/render-markdown-summary.sh) is a
copyable render workflow. The [HTTP request example](../examples/http-run.json)
is for a running `serve` process.

86
docs/config.md Normal file
View File

@@ -0,0 +1,86 @@
# Configuration Reference
This is the canonical reference for Scriptorium application settings. Prompt,
profile, schema, execution-setting, built-in profile, and framework credential
semantics are defined by the
[Promptkit v0.1.0 format reference](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/formats.md).
For command syntax, see the [CLI reference](cli.md); for HTTP request shapes and
outcomes, see the [HTTP API reference](api.md).
## Discovery And Precedence
Application settings are resolved in this order:
1. built-in Scriptorium defaults;
2. a configuration file; then
3. CLI overrides.
When `--config` is omitted, Scriptorium searches
`/usr/local/etc/scriptorium/config.yml` and then `/etc/scriptorium/config.yml`.
If neither exists, it uses built-in defaults. An explicit `--config` path must
exist and decode successfully.
The maintained [minimal configuration](../examples/config.yml) and
[complete configuration](../examples/config.full.yml) are copyable examples.
## Application Configuration File
Configuration is strict YAML: unknown fields are rejected. Empty string values
do not override a prior value. Raw API-key fields are not accepted.
| Field | Default | Meaning |
| --- | --- | --- |
| `prompt_dir` | unset | Promptkit prompt-definition source directory. `run`, `render`, and `serve` require an effective value. |
| `profile_dir` | unset | Optional custom Promptkit profile source directory overlaid on Promptkit built-ins. |
| `schema_dir` | `.` | Promptkit schema source directory for relative schema paths. |
| `server.addr` | `:8080` | Address used by `serve`. |
| `server.artifact_root` | unset | Root that enables HTTP `file` input references. |
| `server.max_request_bytes` | `16777216` | Maximum encoded HTTP request-body bytes; `0` disables the limit. |
| `server.max_artifact_bytes` | `16777216` | Maximum HTTP file-input artifact bytes; `0` disables the limit. |
| `server.max_response_bytes` | `16777216` | Maximum encoded HTTP response bytes; `0` disables the limit. |
| `defaults.render_format` | `text` | Default prepared-run output format: `text` or `json`. |
The size fields must be zero or greater. The [HTTP API](api.md) defines how
each limit is enforced and reported. `server.artifact_root` configures an HTTP
deployment boundary; see [operations](operations.md) for deployment handling.
## Framework Source Mapping
Scriptorium passes `prompt_dir`, `profile_dir`, and `schema_dir` to Promptkit
when constructing its engine. Scriptorium does not redefine or independently
parse those framework file formats.
- Prompt selection, versions, message templates, inputs, output contracts, and
session IDs are Promptkit contracts.
- Profile fields, numeric ranges, execution defaults, overlay precedence,
built-in profiles, and credential rules are Promptkit contracts.
- Schema path behavior and generated-content validation are Promptkit
contracts.
See the
[tagged Promptkit format reference](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/formats.md)
for all of those definitions. The files under
[`examples/prompts`](../examples/prompts/),
[`examples/profiles`](../examples/profiles/), and
[`examples/schemas`](../examples/schemas/) are maintained Scriptorium
application inputs using that tagged format.
## Credentials And Outbound Behavior
Scriptorium maps `--api-key-env` and HTTP `model.api_key_env` into Promptkit
request overrides. Keep secret values in environment variables and store only
their names in configuration or framework source files. Do not place raw keys
in configuration, prompts, profiles, CLI arguments, examples, or HTTP
payloads.
Promptkit's
[OpenAI-compatible integration contract](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/integrations/openai-compatible-chat.md)
defines outbound authentication, provider request mapping, transport limits,
and timeout layering.
## Related References
- [CLI reference](cli.md)
- [HTTP API reference](api.md)
- [Operations guide](operations.md)
- [Promptkit framework formats](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/formats.md)

View File

@@ -1,48 +0,0 @@
# Main `config.yml`
`config.yml` defines application-level defaults used by CLI commands.
By default, Scriptorium looks for `/usr/local/etc/scriptorium/config.yml` and, if not present, then for `/etc/scriptorium/config.yml`. You can also pass `--config PATH`.
## Complete Example
```yaml
prompt_dir: ./prompts
profile_dir: ./profiles
schema_dir: ./schemas
server:
addr: :8080
defaults:
render_format: text
```
## Available Options
- `prompt_dir` (optional): Default directory for prompt definition YAML files.
- `profile_dir` (optional): Default directory for execution profile YAML files.
- `schema_dir` (optional): Base directory for JSON schema files used by `json_schema` validation.
### `server`
- `addr` (optional): HTTP server listen address for `scriptorium serve`.
### `defaults`
- `render_format` (optional): Default output format for `scriptorium render`.
- Allowed values: `text`, `json`.
## Precedence
For run/render/serve settings, precedence is:
1. Explicit CLI flags
2. `config.yml`
3. Built-in defaults
## Notes and Rules
- Unknown YAML fields fail to load (strict decoding).
- This file does not accept API keys.
- `config.yml` sets directory/server defaults only; prompt/profile content remains in their own files.

View File

@@ -1,41 +0,0 @@
# Execution Profile Definitions
Execution Profiles define **how** Scriptorium calls an LLM endpoint.
A profile file is YAML, typically stored under `profiles/`, for example `profiles/local-quality.yaml`.
## Complete Example
```yaml
id: local-quality
endpoint: http://localhost:8000/v1
model: gpt-4.1
temperature: 0.0
max_tokens: 1200
top_p: 1.0
timeout_seconds: 180
reasoning_effort: medium
api_key_env: SCRIPTORIUM_API_KEY
extra_params:
provider: openrouter
route: fallback
```
## Available Options
- `id` (required): Unique profile identifier used by `--profile` or prompt `default_profile`.
- `endpoint` (required): OpenAI-compatible base URL, usually ending in `/v1`.
- `model` (required): Model name to request at that endpoint.
- `temperature` (optional): Sampling temperature. Valid range is `0` to `2`.
- `max_tokens` (optional): Max completion tokens. Must be `>= 0`.
- `top_p` (optional): Nucleus sampling parameter. Valid range is `0` to `1`.
- `timeout_seconds` (optional): Request timeout in seconds. Must be `>= 0`.
- `reasoning_effort` (optional): Provider/model-specific reasoning level string.
- `api_key_env` (optional): Environment variable name that holds the API key.
- `extra_params` (optional): String key/value map for provider-specific parameters.
## Notes and Rules
- Raw API keys are not supported. Do **not** add `api_key` fields.
- Unknown YAML fields fail to load (strict decoding).
- If `api_key_env` is set, the environment variable must be present when the run executes.

View File

@@ -1,73 +0,0 @@
# Prompt Definition Files
Prompt Definitions define **what** Scriptorium should do.
A prompt file is YAML, typically stored under `prompts/`, for example `prompts/generic.structured_events.yaml`.
## Complete Example
```yaml
id: generic.structured_events
version: "1.0.0"
default_profile: local-quality
description: Extract events from a transcript into structured JSON.
inputs:
- name: transcript
required: true
content_type: text/markdown
description: Source transcript
- name: glossary
required: false
content_type: text/yaml
description: Optional glossary context
messages:
- role: system
content: |
You are a structured extraction assistant.
Return only JSON.
- role: user
content_file: ./generic.structured_events.user.md
output:
format: json
validation_mode: json_schema
schema_path: structured_events.schema.json
repair_attempts: 1
```
## Available Options
- `id` (required): Prompt identifier used by `--prompt` / `prompt_id`.
- `version` (required): Prompt version string.
- `default_profile` (optional): Execution profile ID used when no explicit profile is provided.
- `description` (optional): Human-readable description.
### `inputs[]`
- `name` (required): Logical input name referenced in templates via `{{input "name"}}`.
- `required` (optional): If `true`, run fails when input is missing.
- `content_type` (optional): Metadata only (not enforced yet).
- `description` (optional): Human-readable input description.
### `messages[]`
- `role` (required): Message role such as `system` or `user`.
- `content` (optional): Inline Go-template message body.
- `content_file` (optional): Path to a template file.
Each message must set **exactly one** of `content` or `content_file`.
### `output`
- `format` (required): One of `text`, `markdown`, `json`.
- `validation_mode` (required): One of `none`, `basic`, `json`, `json_schema`.
- `schema_path` (required when `validation_mode: json_schema`): Path to JSON Schema file.
- `repair_attempts` (required): Number of bounded repair retries (`>= 0`).
## Notes and Rules
- Unknown YAML fields fail to load (strict decoding).
- `content_file` paths are resolved relative to the prompt YAML file.
- For `json_schema` validation mode, Scriptorium also sends provider-level structured output requests automatically.

View File

@@ -1,82 +0,0 @@
# JSON Schema Definition Files
Schema definition files describe the expected JSON output contract for prompts that use:
- `output.format: json`
- `output.validation_mode: json_schema`
Schema files are JSON, typically stored under `schemas/`, for example `schemas/structured_events.schema.json`.
## Complete Example
```json
{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"$id": "https://example.com/schemas/structured-events.schema.json",
"title": "Structured Events",
"description": "Expected shape for extracted event output",
"type": "object",
"properties": {
"summary": {
"type": "string",
"minLength": 1,
"description": "High-level session summary"
},
"events": {
"type": "array",
"items": {
"type": "object",
"properties": {
"title": { "type": "string" },
"type": {
"type": "string",
"enum": ["discovery", "combat", "social", "travel", "downtime", "other"]
},
"notes": { "type": "string" }
},
"required": ["title", "type"],
"additionalProperties": false
},
"minItems": 0
}
},
"required": ["summary", "events"],
"additionalProperties": false,
"$defs": {
"nonEmptyString": {
"type": "string",
"minLength": 1
}
}
}
```
## Available Options
Scriptorium does not define custom schema keywords. It expects a valid JSON Schema document and passes it to the validator/provider.
Commonly used JSON Schema options include:
- `$schema`: Draft identifier URI.
- `$id`: Schema identifier URI.
- `title`: Human-readable schema title.
- `description`: Human-readable schema description.
- `type`: Expected JSON type (`object`, `array`, `string`, etc.).
- `properties`: Object field definitions.
- `required`: Required object fields.
- `additionalProperties`: Whether undeclared fields are allowed.
- `items`: Array item schema.
- `enum`: Allowed literal values.
- `const`: Single allowed literal value.
- `oneOf`, `anyOf`, `allOf`: Composition rules.
- `minimum`, `maximum`: Numeric bounds.
- `minLength`, `maxLength`, `pattern`: String constraints.
- `minItems`, `maxItems`: Array constraints.
- `$defs`: Reusable local definitions.
- `$ref`: Reference to another schema/definition.
## Notes and Rules
- Schema path comes from prompt `output.schema_path` and is resolved relative to `schema_dir`.
- If schema loading fails for `json_schema` mode, the run fails before the LLM request.
- Keep schemas strict (`additionalProperties: false`) when you want predictable output shape.

38
docs/consumers/api.md Normal file
View File

@@ -0,0 +1,38 @@
# Consumer Integration Overview
Scriptorium exposes executable interfaces. Choose between a local subprocess
and the HTTP service according to the boundary your application needs.
| Interface | Use when |
| --- | --- |
| CLI subprocess | The consumer needs a synchronous local process boundary or prepared output. |
| HTTP API | The consumer needs a service boundary or remote access. |
- CLI subprocess: [subprocess integration](../integrations/subprocess.md)
- HTTP service: [HTTP API reference](../api.md)
- Application configuration: [configuration reference](../config.md)
Go applications that need an in-process prompt framework should import
Promptkit directly. The tagged
[Promptkit Go consumer guide](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/consumers/pkg-promptkit.md)
owns that interface; Scriptorium does not provide a Go library package.
Consumers arriving from the former Scriptorium Go API should follow the
[migration guide](migrating-to-promptkit.md).
## Consumer Responsibilities
Consumers are responsible for:
- selecting and deploying prompt, profile, and schema assets;
- supplying required inputs and template variables;
- supplying credentials through the chosen interface;
- protecting rendered prompts and generated artifacts as potentially
sensitive;
- deciding whether validation-failed output is usable; and
- retrying only when another model call is acceptable.
Scriptorium does not persist run state. A retry can produce different output
and can incur another provider request. CLI exits belong to the
[CLI reference](../cli.md), HTTP status behavior belongs to the
[HTTP API reference](../api.md), and framework semantics belong to
[Promptkit v0.1.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/formats.md).

View File

@@ -0,0 +1,109 @@
# Migrate From Scriptorium To Promptkit
## Supported Migration Boundary
Scriptorium `v0.11.1` at
`gitea.maximumdirect.net/eric/scriptorium` is the final release that provides
the former in-process Go framework. Promptkit `v0.1.0` at
`gitea.maximumdirect.net/eric/promptkit` is the destination for that framework
API. Scriptorium `v0.12.0` and later provide the CLI and HTTP application only.
There is no Scriptorium compatibility facade, alias package, forwarding
package, or deprecated wrapper. A consumer that cannot migrate may remain
pinned to Scriptorium `v0.11.1`, but that framework-bearing line does not
provide the slim application release.
## Update A Go Consumer
Start from a clean consumer checkout and review the pending diff before
committing it. Add the published Promptkit module:
```sh
go get gitea.maximumdirect.net/eric/promptkit@v0.1.0
```
For an ordinary consumer that imports the former root package under its
default name, replace the exact import and package qualifier, then format the
changed Go files:
```sh
git grep -l \
'"gitea.maximumdirect.net/eric/scriptorium"' \
-- '*.go' |
while IFS= read -r go_file
do
perl -pi -e \
's{"gitea.maximumdirect.net/eric/scriptorium"}{"gitea.maximumdirect.net/eric/promptkit"}g; s{\bscriptorium\.}{promptkit.}g' \
"$go_file"
gofmt -w "$go_file"
done
```
Inspect the resulting diff. Consumers that used an import alias should retain
or deliberately rename that alias instead of applying the qualifier
replacement mechanically.
Remove the now-unused Scriptorium requirement through module tidiness and run
the consumer's complete tests:
```sh
go mod tidy
go test ./...
```
Confirm that `go.mod` selects Promptkit `v0.1.0` and that no Go file imports
the former Scriptorium package:
```sh
test "$(
go list -m -f '{{.Path}}@{{.Version}}' \
gitea.maximumdirect.net/eric/promptkit
)" = 'gitea.maximumdirect.net/eric/promptkit@v0.1.0'
if git grep -n \
'gitea.maximumdirect.net/eric/scriptorium' \
-- '*.go'
then
printf '%s\n' 'a former Scriptorium Go import remains' >&2
exit 1
fi
```
## Compatibility And Additions
Promptkit preserves the established engine, request, result, profile,
source-option, model-client, artifact, validation-value, and public-error
shapes where practical. Exact declarations and current behavior belong to the
tagged [Promptkit consumer guide](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/consumers/pkg-promptkit.md)
and Go source.
Promptkit also includes migration-relevant public contracts that were not in
Scriptorium `v0.11.1`:
- [`WithArtifactReader`](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/engine.go#L96-L105)
and the
[`ArtifactReader` declaration](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/types.go#L129-L135)
provide the artifact-reading extension described by the tagged
[extension-interface guide](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/consumers/pkg-promptkit.md#extension-interfaces).
- [`ErrProfileRequired` and `ErrAPIKeyEnvMissing`](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/engine.go#L28-L40)
provide the specific identities described by the tagged
[error guide](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/consumers/pkg-promptkit.md#errors).
Use those tagged owners for exact signatures, wrapping guarantees, and
extension behavior.
## Verify Consumer Behavior
Source compatibility is only the first check. Exercise the behavior the
consumer actually relies upon, especially:
- prompt, profile, and schema source selection;
- direct and environment-based credentials;
- caller, generation, and transport timeout layering;
- output validation and validation-failure handling;
- injected model-client and artifact-reader extensions; and
- every `errors.Is` branch used for recovery or classification.
Also verify any serialized values, redaction expectations, filesystem policy,
and provider integration behavior that crosses the consumer's own boundary.
Promptkit owns the in-process framework contract; Scriptorium owns only its
executable CLI and HTTP application interfaces.

56
docs/development.md Normal file
View File

@@ -0,0 +1,56 @@
# Development
This is the contributor entry point for Scriptorium. Scriptorium is an
application that consumes the public
[Promptkit v0.1.0 package](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/consumers/pkg-promptkit.md);
framework implementation work belongs in Promptkit.
## Initial Orientation
Before starting work:
1. inspect the working tree and preserve unrelated changes;
2. read the [architecture policy](policy/architecture.md);
3. follow the task-specific contracts and internal documents below; and
4. inspect the relevant implementation and tests before changing them.
Also read the [documentation policy](policy/documentation.md) before changing
documentation and the [testing policy](policy/testing.md) before changing
tests.
## Task-Specific Reading Guide
| Task | Read before changing |
| --- | --- |
| Repository orientation or component responsibility | [Internal component overview](internal/overview.md) and [architecture policy](policy/architecture.md) |
| CLI commands, flags, output, or exit behavior | [CLI contract](cli.md) and [adapter internals](internal/adapters.md) |
| HTTP routes, DTOs, limits, status mapping, or artifact policy | [HTTP API contract](api.md), [adapter internals](internal/adapters.md), and [source internals](internal/sources.md) |
| Application configuration or precedence | [Configuration contract](config.md), [adapter internals](internal/adapters.md), and [source internals](internal/sources.md) |
| Prepared-run presentation | [CLI contract](cli.md), [adapter internals](internal/adapters.md), and `internal/format` |
| Prompt, profile, schema, generation, or validation semantics | [Promptkit framework formats](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/formats.md) and the [Promptkit consumer guide](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/consumers/pkg-promptkit.md) |
| OpenAI-compatible outbound behavior or timeout layering | [Promptkit integration contract](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/integrations/openai-compatible-chat.md) |
| Subprocess behavior | [Subprocess integration](integrations/subprocess.md) and [CLI contract](cli.md) |
| Runtime operation or recovery | [Operations](operations.md) |
| Release packaging or publication | The [release procedure](release.md), [hosted release workflow](../.woodpecker/release.yml), and [architecture policy](policy/architecture.md) |
| Examples or copyable assets | The owning Scriptorium contract, the relevant [Promptkit format contract](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/formats.md), and the related files under `examples/` |
| Architecture decisions or future work | The [documentation policy](policy/documentation.md), relevant accepted ADRs, and relevant roadmap documents |
Cross-project changes land and release in Promptkit before Scriptorium adopts
the tagged version. Do not commit a Go workspace, local replacement, vendored
Promptkit source, or an import of a Promptkit `internal` package.
## Baseline Validation
For code changes, run:
```bash
go test ./...
go test -race ./...
go vet ./...
go build ./cmd/scriptorium
```
Check formatting with `gofmt`, run `git diff --check`, and validate affected
examples and documentation links. Documentation-only work does not require
unrelated new tests, but commands and examples changed by documentation must be
run.

View File

@@ -1,322 +0,0 @@
# Narratio -> Scriptorium CLI Integration
## 1. Purpose
This document defines how Narratio should invoke Scriptorium through the **public CLI**.
This is a **subprocess integration contract**, not an internal Go API contract.
## 2. Assumptions
- `scriptorium` is installed and available on `PATH`.
- Scriptorium is configured with `config.yml`.
- `config.yml` provides `prompt_dir`, `profile_dir`, and `schema_dir` as needed.
- Prompt and profile libraries are already deployed for the environment.
- Narratio provides prepared artifact files (for example polished transcript, glossary, previous recap, campaign notes).
- Initial integration is synchronous subprocess execution.
- Narratio remains the orchestrator.
In normal operation, Narratio does not need to pass `--prompt-dir` and `--profile-dir` if they are supplied by Scriptorium config.
Narratio may pass `--config <PATH>` when it must use a non-default Scriptorium config file.
## 3. Core Commands Narratio May Call
Primary commands for subprocess integration:
- `scriptorium run`
- `scriptorium render`
For production generation, use `scriptorium run`.
`scriptorium render` is for debugging, dry-runs, test assertions, and validating command construction without LLM execution.
Note: `scriptorium serve` and HTTP API exist, but they are not the initial integration path.
## 4. Command Selection Guidance
- Use `run` to generate an output artifact.
- Use `render` to inspect the prepared prompt and effective settings without calling the LLM.
- Use `render --format json` when Narratio/tests need structured prepare output.
## 5. Recommended `run` Invocation Shape
Production shape:
```bash
scriptorium run \
--prompt <prompt_id> \
--input transcript=<processed-transcript-path> \
--out <output-artifact-path>
```
Common optional additions:
- `--config <path>`: use a specific Scriptorium config file.
- `--profile <profile_id>`: override prompt default profile.
- `--var name=value` (repeatable): small metadata values.
- `--input name=path` (repeatable): additional named artifacts.
- `--timeout <duration>`: per-run timeout override.
- Runtime model override flags (`--llm-base-url`, `--model`, etc.) only for exceptional/operator-directed cases.
## 6. Recommended `render` Invocation Shape
Human-readable debug shape:
```bash
scriptorium render \
--prompt <prompt_id> \
--input transcript=<processed-transcript-path> \
--format text
```
Structured debug/test shape:
```bash
scriptorium render \
--prompt <prompt_id> \
--input transcript=<processed-transcript-path> \
--format json \
--out <render-debug-path>
```
`render` does **not** call the LLM, does **not** validate model output, and does **not** perform repair.
## 7. Inputs
- Pass inputs as repeated `--input name=path` flags.
- `name` must match the Prompt Definition input name.
- Prefer absolute paths, or paths relative to a working directory controlled by Narratio.
- Pass Audita output as the primary transcript input.
- Additional inputs may include glossary, previous recap, campaign notes, event logs, final state maps, or other prompt-specific artifacts.
- Scriptorium reads input files directly; Narratio does not need to inline file content for CLI use.
## 8. Variables
Use repeated `--var name=value` for small metadata values.
Typical examples:
- `session_date`
- `session_id`
- `campaign_name`
- `previous_session_id`
- `output_kind`
Large content belongs in input files, not `--var` values.
## 9. Prompt IDs and Output Artifact Types
Narratio should treat prompt IDs as configuration, not hardcoded business logic.
Narratio config may map stage/output names to prompt IDs, for example:
- session recap prompt
- structured event extraction prompt
- glossary suggestion prompt
- player-facing summary prompt
Prompt IDs used by Narratio should come from the deployed Scriptorium prompt library.
## 10. Profiles
- Prompts may declare `default_profile`.
- Narratio may omit `--profile` to use prompt default profile.
- Narratio may pass `--profile` to force profile selection.
- This enables environment/profile selection like `local-fast`, `local-quality`, `frontier`, `batch`, or test profiles.
- Profile names should generally be Narratio configuration values.
## 11. Runtime Overrides
Supported runtime override flags:
- `--llm-base-url`
- `--model`
- `--api-key-env`
- `--temperature`
- `--max-tokens`
- `--top-p`
- `--timeout`
Guidance:
- Keep normal model/runtime settings in Execution Profiles.
- Use runtime overrides only for explicit per-run exceptions, tests, or operator overrides.
- Never pass raw API keys on the command line.
- `--api-key-env` names an environment variable; Narratio must ensure that variable is set in subprocess environment.
## 12. Config Behavior
- Default config path: `/etc/scriptorium/config.yml`.
- `--config <PATH>` overrides default path.
- Missing default config is allowed by Scriptorium.
- If `--config` is provided explicitly, the file must exist and be valid.
- CLI flags override `config.yml`.
- `config.yml` overrides built-in application defaults.
Narratio can either:
- rely on system default config path, or
- carry an explicit config path and pass `--config`.
## 13. Environment Handling
Subprocess environment recommendations:
- Pass through required API-key environment variables referenced by `api_key_env`.
- Do not pass raw API keys as CLI arguments.
- Avoid logging full environment dumps.
- Capture stdout and stderr separately.
- Use a controlled working directory.
- Prefer absolute artifact paths.
## 14. Output Handling
For `scriptorium run`:
- Use `--out` when Narratio needs durable artifact files.
- Without `--out`, artifact content is written to stdout.
- Preferred orchestration pattern: always use `--out`, then treat the file as stage output artifact.
- Capture stderr for diagnostics.
For `scriptorium render`:
- Use `--out` to store render diagnostics.
- Use `--format json` when tests need to inspect selected profile, effective runtime settings, input hashes, prompt hash, and rendered messages.
## 15. Exit Status and Errors
Current CLI behavior (verified from implementation/tests):
- `0`: success.
- `1`: runtime/parse/config/load/render/generation/IO error.
- `2`: run completed but output validation failed (`ValidationFailed`).
Additional details:
- On `run`, output artifact write happens before exit code selection. If validation fails, artifact may still be written and exit code is `2`.
- `stderr` carries both errors and normal run summary output; non-empty stderr alone does not imply failure.
- `render` returns `0` on success and `1` on failures.
Narratio should treat non-zero exit codes as failed stage execution, but may record generated artifact paths if a run exited `2` and output file exists.
## 16. Recommended Narratio Integration Pattern
1. Build CLI args from Narratio stage configuration.
2. Use subprocess context cancellation/timeout.
3. Pass absolute input paths.
4. Pass `--out` to a session-scoped artifact path.
5. Add `--var` metadata values.
6. Optionally add `--config`.
7. Optionally add `--profile`.
8. Ensure required API-key env vars are present.
9. Run subprocess synchronously.
10. Capture stdout/stderr separately.
11. On success, store output artifact path and invocation metadata in stage artifacts.
12. On failure, store exit code and stderr diagnostics in stage status.
## 17. Suggested Narratio Configuration Shape
Illustrative (not required schema):
```yaml
scriptorium:
config_path: /etc/scriptorium/config.yml
stages:
session_recap:
prompt_id: dnd.session_recap
profile_id: local-quality # optional
inputs: [transcript, glossary, previous_recap]
vars: [session_id, session_date, campaign_name]
output_path_template: artifacts/{session_id}/session_recap.md
timeout: 2m
render_debug: false
```
The key idea: map Narratio stage/artifact names to prompt ID, optional profile, expected inputs, and output destination.
## 18. Testing Strategy for Narratio Integration
- Use `scriptorium render --format json` to verify command construction without LLM calls.
- Use dedicated test prompt/profile libraries for integration tests.
- Use small fixture transcripts.
- Verify missing-input failure behavior.
- Verify prompt `default_profile` behavior.
- Verify explicit `--profile` override behavior.
- Verify `--config` behavior (default and explicit).
- Verify output file creation when `--out` is used.
- Verify stderr capture on failures.
- Avoid real API keys in tests.
## 19. Security and Privacy Notes
- Never pass raw API keys on command line.
- Do not log full rendered prompts by default; transcripts may contain sensitive content.
- Avoid logging prompt content unless explicit debug mode is enabled.
- Treat generated artifacts as potentially sensitive.
- Use session-scoped, access-controlled output paths.
- `api_key_env` names should come from environment management, not embedded secrets.
## 20. Initial D&D Artifact Generation Examples
These are examples only. Use prompt IDs from the deployed prompt library.
Session recap:
```bash
scriptorium run \
--prompt dnd.session_recap \
--input transcript=/work/session-42/transcript.polished.md \
--input glossary=/work/session-42/glossary.yml \
--out /work/session-42/artifacts/session_recap.md
```
Structured events:
```bash
scriptorium run \
--prompt dnd.structured_events \
--input transcript=/work/session-42/transcript.polished.md \
--out /work/session-42/artifacts/structured_events.json
```
Glossary suggestions:
```bash
scriptorium run \
--prompt dnd.glossary_suggestions \
--input transcript=/work/session-42/transcript.polished.md \
--input previous_recap=/work/session-41/artifacts/session_recap.md \
--out /work/session-42/artifacts/glossary_suggestions.md
```
Player-facing summary:
```bash
scriptorium run \
--prompt dnd.player_summary \
--input transcript=/work/session-42/transcript.polished.md \
--input structured_events=/work/session-42/artifacts/structured_events.json \
--out /work/session-42/artifacts/player_summary.md
```
## 21. Non-Goals
Initial Narratio integration should not:
- call Scriptorium internal Go packages
- use HTTP API as the primary path
- expect Scriptorium to read S3 refs directly
- make Scriptorium responsible for Narratio stage state
- make Scriptorium responsible for notification
- require Scriptorium to understand D&D workflow semantics beyond prompt definitions
## 22. Future Extension Notes
Possible later extensions:
- HTTP API integration
- S3 artifact references if Scriptorium adds S3 reader support
- storing render diagnostics alongside generated artifacts
- token budgeting/prompt-size checks
- batch execution if Scriptorium later adds batch support

View File

@@ -0,0 +1,41 @@
# Subprocess Integration
This document covers process-boundary behavior for callers that invoke
Scriptorium as a child process. Command syntax, flags, output, and exit codes
are defined by the [CLI reference](../cli.md). Interface selection belongs in
the [consumer integration overview](../consumers/api.md).
## Process Contract
Use `scriptorium render` when the caller needs prepared output without a model
call, and `scriptorium run` for generation. Pass an explicit `--config` or
make the configuration search paths available to the child process; configuration
discovery, fields, profile selection, and credential mechanisms are defined in
the [configuration reference](../config.md).
Pass required API-key environment variables through the child environment. Do
not place raw API keys in arguments. Keep the environment limited to the values
needed for the selected profile.
## Streams And Output Ownership
Capture stdout and stderr separately. Stdout contains the requested artifact or
prepared output unless the caller selects an output file; stderr contains
summaries, diagnostics, and server messages. The exact destinations and status
meanings are part of the [CLI reference](../cli.md), not a stable stderr data
protocol.
When using `--out`, the caller owns the output path, its permissions, and
cleanup. Treat rendered prompts, generated artifacts, stdout, and stderr as
potentially sensitive.
## Cancellation And Recovery
A CLI invocation performs one synchronous request and creates no durable run
state. A supervising process that needs cancellation must terminate the child
process according to its own process-management policy. A later invocation is a
new request and can make another model call; there is no resume or checkpoint
protocol.
For deployment, filesystem permissions, and sensitive-artifact handling, see
the [operations guide](../operations.md).

87
docs/internal/adapters.md Normal file
View File

@@ -0,0 +1,87 @@
# Adapter Internals
## Purpose
Scriptorium adapters translate executable inputs into Promptkit public requests
and translate Promptkit results or errors back to CLI or HTTP behavior. They
own IO and presentation mechanics, not framework decisions.
External contracts are canonical in the [CLI reference](../cli.md) and
[HTTP API reference](../api.md). Promptkit's public engine contract is
described by its tagged
[Go consumer guide](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/consumers/pkg-promptkit.md).
## Components And Collaborators
- `cmd/scriptorium` passes process arguments and streams to
`internal/adapter/cli`.
- `internal/adapter/cli` resolves settings through `internal/config`,
constructs `promptkit.Engine`, maps CLI values to `promptkit.RunRequest`,
and owns output files, summaries, and exit codes.
- `internal/adapter/http` strictly decodes request DTOs, maps them to Promptkit
public values, calls its adapter-owned `Runner` interface, and maps results
and errors to HTTP DTOs.
- `internal/format` renders `promptkit.PreparedRun` values as deterministic text
or JSON.
## Wiring Flows
### CLI
`run` calls `promptkit.Engine.Run`; `render` calls
`promptkit.Engine.Prepare`. Both share request mapping for prompt/profile
selection, file inputs, variables, and presence-aware execution overrides.
Omitted framework settings remain zero values so Promptkit resolves its own
defaults.
`serve` constructs Scriptorium's restricted HTTP artifact reader, injects it
with `promptkit.WithArtifactReader`, passes the engine through the HTTP
adapter's consumer-owned `Runner` interface, and starts the server.
### HTTP
The handler enforces transport limits and strict JSON decoding before mapping
DTOs into `promptkit.RunRequest`, `promptkit.ArtifactRef`, and
`promptkit.ExecutionTargetOverride`. On success it reads Promptkit artifact,
validation, model, usage, and metadata values directly.
Failure mapping uses `errors.Is` against Promptkit's public sentinels and the
HTTP reader's Scriptorium-owned containment and size errors. Wrapped reader
errors preserve their identity through Promptkit's artifact-load boundary.
## Package-Local Guarantees
- Adapters contain no copied framework types or orchestration.
- Configuration is resolved before Promptkit engine construction.
- Explicit numeric overrides preserve presence, including zero.
- HTTP DTO and error mapping remains stable and transport-owned.
- Resolved secrets are not serialized or printed.
- No adapter creates durable run state.
## Verification
Inspect:
- `internal/adapter/cli/run_test.go`
- `internal/adapter/http/handler_test.go`
- `internal/adapter/http/artifact_reader_test.go`
- `internal/format/prepared_run_test.go`
- `internal/adapter/dependency_test.go`
The adapter tests protect parsing, configuration mapping, output, status
mapping, restricted artifacts, and representative real Promptkit-engine
workflows. The dependency test protects the repository boundary.
## Change Recipes
For a CLI or HTTP change:
1. identify the Scriptorium-owned external contract;
2. map through Promptkit public values without copying framework semantics;
3. add or update the narrow application-owned test;
4. update the canonical Scriptorium contract; and
5. coordinate and tag Promptkit first if a required public capability is
genuinely absent.
Update [source internals](sources.md) when application source locations or HTTP
artifact containment changes.

18
docs/internal/overview.md Normal file
View File

@@ -0,0 +1,18 @@
# Internal Component Overview
This is the complete inventory of Scriptorium's implemented Go components.
The [architecture policy](../policy/architecture.md) owns normative boundaries;
public behavior belongs in the linked contracts.
| Component | Implemented responsibility | References |
| --- | --- | --- |
| `cmd/scriptorium` | Process entrypoint that delegates arguments and streams to the CLI adapter. | [CLI contract](../cli.md), [adapter internals](adapters.md) |
| `internal/adapter/cli` | Parses commands, resolves application settings, constructs Promptkit engines, maps requests, and owns process output and exit behavior. | [CLI contract](../cli.md), [adapter internals](adapters.md) |
| `internal/adapter/http` | Owns routes, DTOs, strict decoding, limits, Promptkit request/result mapping, public error mapping, and restricted HTTP artifact reading. | [HTTP API](../api.md), [adapter internals](adapters.md), [source internals](sources.md) |
| `internal/config` | Discovers and strictly decodes application configuration and applies built-in and CLI precedence. | [configuration contract](../config.md), [adapter internals](adapters.md) |
| `internal/defaults` | Holds Scriptorium-owned application and HTTP defaults. | [configuration contract](../config.md) |
| `internal/format` | Formats Promptkit prepared-run values for CLI text or JSON output. | [CLI contract](../cli.md), [adapter internals](adapters.md) |
Framework implementation packages are provided by
[Promptkit v0.1.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/consumers/pkg-promptkit.md)
and are not part of this repository.

68
docs/internal/sources.md Normal file
View File

@@ -0,0 +1,68 @@
# Source Internals
## Purpose
This document covers Scriptorium-owned source locations and the restricted HTTP
artifact reader. Prompt, profile, schema, and ordinary artifact semantics are
owned by the tagged
[Promptkit format reference](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/formats.md).
## Application Source Locations
`internal/config` resolves `prompt_dir`, `profile_dir`, and `schema_dir` from
Scriptorium defaults, configuration files, and CLI overrides.
`internal/adapter/cli` passes those paths into `promptkit.Config` when
constructing the engine.
Scriptorium does not search, parse, validate, or overlay framework source files
itself. Promptkit owns prompt selection, profile built-ins and overlays, schema
resolution, ordinary file artifacts, and the related error identities.
The [configuration reference](../config.md) owns Scriptorium's source-location
fields and precedence. Maintained files under `examples/` are application
inputs that use Promptkit's tagged formats.
## Restricted HTTP Artifact Reader
`internal/adapter/http` implements `promptkit.ArtifactReader` for HTTP
requests. The `serve` path injects it with
`promptkit.WithArtifactReader`, replacing Promptkit's ordinary reader for
inbound HTTP inputs.
The reader:
- accepts inline references without an artifact root;
- denies file references when no root is configured;
- resolves relative paths below the configured root;
- accepts absolute paths only when they are lexically within that root;
- rejects lexical traversal outside the root;
- applies the configured file byte limit, with zero meaning unlimited;
- preserves content type, body, size, hash, name, and URI metadata; and
- honors context cancellation.
Containment is lexical and does not resolve symlinks. The operating system
follows symlinks after the check. The [HTTP API](../api.md) owns observable
request outcomes, and [operations](../operations.md) owns safe deployment
permissions and root selection.
Reader errors remain identifiable after Promptkit wraps them as artifact-load
failures, allowing the HTTP adapter to preserve Scriptorium status and error
codes.
## Verification And Change Recipe
Inspect:
- `internal/config/config_test.go`
- `internal/adapter/cli/run_test.go`
- `internal/adapter/http/artifact_reader_test.go`
- `internal/adapter/http/handler_test.go`
When changing an application source location or HTTP artifact policy:
1. preserve strict configuration precedence and the Promptkit public boundary;
2. keep containment and size policy in Scriptorium;
3. update focused configuration, reader, and handler tests;
4. update the [configuration](../config.md), [HTTP](../api.md), and
[operations](../operations.md) contracts as applicable; and
5. do not duplicate Promptkit loaders, formats, or ordinary artifact behavior.

160
docs/operations.md Normal file
View File

@@ -0,0 +1,160 @@
# Operations Guide
## Scope And References
This runbook covers deployment, normal operation, capacity planning, and safe
recovery for Scriptorium. It does not redefine invocation syntax, configuration
fields, or HTTP wire behavior.
- [CLI reference](cli.md): commands, output destinations, and exit codes.
- [Configuration reference](config.md): application settings, source
locations, defaults, and credential mapping.
- [Promptkit framework formats](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/formats.md):
prompt, profile, schema, execution-setting, and framework credential
contracts.
- [HTTP API reference](api.md): route, request/response schema, status codes,
limits, and HTTP artifact access.
- [Consumer integration overview](consumers/api.md): caller responsibilities.
## Operational Model And State
Scriptorium handles one prompt request for each CLI invocation or HTTP request.
It has no durable run store, archive, checkpoint, cache, or resume mechanism.
A failed or interrupted request is recovered by correcting its inputs,
configuration, or environment and submitting a new request.
Generated artifacts, rendered prompts, model output, and run metadata are
caller-owned data. Retention, encryption, backup, and deletion are deployment
responsibilities.
## Deploy The Filesystem And Process
Provide the process with readable configured Promptkit prompt, profile, and
schema sources that follow the tagged framework formats. For an HTTP deployment
that accepts file artifacts, use a dedicated, narrow artifact directory rather
than a general-purpose or sensitive filesystem tree.
Run Scriptorium under an identity that can:
- read only the prompt, profile, schema, and allowed input-artifact paths it
needs;
- read the required credential environment variables without writing them to
files or logs; and
- write only caller-selected output locations when CLI output files are used.
Do not make the HTTP artifact directory writable by untrusted users. The HTTP
artifact containment behavior is lexical and the operating system follows
symlinks; account for that when choosing ownership and mount boundaries. See
the [HTTP API reference](api.md) for the externally observable behavior.
## Supply Credentials And Protect Runtime Data
Set secret values in the process environment and configure only their
environment-variable names. Do not put raw keys in configuration, prompt or
profile files, process arguments, HTTP payloads, captured command lines, or
debug dumps.
Treat stdout, stderr, prepared-run output, generated artifacts, and HTTP
responses as potentially sensitive. Send service logs to a controlled collector
and apply the same retention and access rules as for model input and output.
## Run A Normal Workflow
Before changing production inputs, profiles, or schemas:
1. confirm the deployed configuration selects the intended sources and model
credentials;
2. use [`render`](cli.md) with the same request inputs and variables to confirm
preparation without a model call;
3. use [`run`](cli.md) for generation; and
4. retain or discard validation-failed output according to the caller's
policy.
The [maintained render script](../examples/render-markdown-summary.sh) is a
copyable preflight example. The CLI reference owns its complete invocation and
exit semantics.
## Expose The HTTP Service
The HTTP service has no built-in authentication or authorization. Place it on a
trusted network or behind an authenticated reverse proxy, API gateway, or
equivalent access control. Restrict who can reach it and who can read the
artifact root.
Use a service manager or supervisor appropriate to the deployment to manage
process lifetime, restart policy, log capture, and environment injection. The
[HTTP API reference](api.md) owns client request shapes, status behavior, and
artifact-access outcomes.
## Plan Capacity And Limits
Capacity is primarily determined by concurrent model calls, input and output
sizes, schema complexity, provider latency, and network behavior. Size limits
protect request bodies, HTTP file artifacts, and encoded responses; configure
them through the [configuration reference](config.md) and rely on the
[HTTP API reference](api.md) for their response effects.
Before increasing a limit:
1. measure representative input, generated-output, and optional raw-output
sizes;
2. confirm memory, network, and upstream-provider capacity;
3. retain an upstream request-size and authentication boundary; and
4. test the intended workload in a non-production environment.
For large local inputs, prefer a controlled file-artifact directory over
placing arbitrary paths on the service host. Avoid disabling a limit unless an
equivalent trusted control exists elsewhere.
## Diagnose And Recover
### Preparation Or Configuration Failure
Capture the CLI diagnostic or HTTP error response, then verify the selected
configuration, prompt ID, profile selection, source readability, and input
mapping. Use `render` with the same request when it is unclear whether failure
occurs before model execution. Consult the [CLI reference](cli.md), the
[configuration reference](config.md), and the [HTTP API reference](api.md) for
the exact interface contract.
### Credential Or Provider Failure
Confirm that the process environment contains the configured credential name
without printing the secret. Check endpoint reachability and provider health
from the process network. If preparation succeeds but generation fails, inspect
the selected model settings in prepared output and the service's controlled
logs. Correct the deployment or provider issue, then submit a new request.
### Artifact Or Permission Failure
Verify that the process can read the intended local input. For HTTP file
artifacts, verify the deployment's artifact root, ownership, path layout, and
file size. Do not widen filesystem permissions or the allowed root merely to
make an arbitrary path work; move or copy the required artifact into the
controlled location instead.
### Validation Failure
A generated-content validation failure is distinct from a runtime failure.
CLI `run` reports the validation result and error count in its success summary;
it does not print the individual validation messages. For HTTP, inspect the
validation object in the response according to the [HTTP API reference](api.md).
Use rendered input and generated output to determine whether prompt instructions,
the selected model, or the schema needs correction. If schema loading or
compilation itself fails, correct the source deployment or schema document
before rerunning.
### HTTP Limit Or Request Failure
Compare the request, artifact, or expected response size with the deployed
configuration, and validate the request against the [HTTP API reference](api.md).
Reduce the payload, use an appropriate controlled artifact source, omit
unneeded raw output, or adjust the deployment limit after capacity review.
## Cleanup And Reruns
Because no run state is retained, cleanup concerns caller-owned output files,
logs, and artifacts only. Remove or rotate them using the deployment's normal
retention policy. After a correction, rerun the request from the beginning;
there is no safe resume point.

102
docs/policy/architecture.md Normal file
View File

@@ -0,0 +1,102 @@
# Architecture
This document defines Scriptorium's current application architecture and
durable development boundaries.
## System Shape
Scriptorium is an executable application with three entry paths: CLI `run`, CLI
`render`, and the HTTP service started by `serve`. It does not expose a reusable
root Go package.
The application consumes
[Promptkit v0.1.0](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/consumers/pkg-promptkit.md)
through its supported root package. Promptkit owns prompt execution,
preparation, source formats, built-in profiles, model-client behavior, and
validation. Scriptorium owns application configuration, executable adapters,
prepared-run presentation, process behavior, and HTTP deployment policy.
The concrete package inventory is maintained in the
[internal overview](../internal/overview.md).
## Dependency Direction
```text
cmd/scriptorium
|
v
CLI and HTTP adapters, configuration, defaults, and formatting
|
v
gitea.maximumdirect.net/eric/promptkit
```
- Retained application packages may import Promptkit's root package.
- They must not import Promptkit `internal` packages.
- They must not import the removed Scriptorium root facade or recreate former
framework package families.
- Adapter-owned interfaces use Promptkit public values when a consumer-side
substitution boundary is needed.
- Scriptorium passes omitted framework settings as zero values so Promptkit
applies its own defaults.
The repository architecture guard enforces these import and removal
invariants.
## Retained Boundaries
- `internal/adapter/cli` owns commands, flags, configuration precedence,
process streams, output files, summaries, and exit codes.
- `internal/adapter/http` owns routes, strict JSON DTOs, size limits, response
mapping, status mapping, and the restricted artifact reader.
- `internal/config` owns discovery and strict decoding of Scriptorium
application configuration.
- `internal/defaults` owns Scriptorium application and HTTP defaults only.
- `internal/format` owns deterministic prepared-run text and JSON presentation.
- Promptkit owns framework orchestration and contracts. Its
[format reference](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/formats.md)
and
[outbound integration contract](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/integrations/openai-compatible-chat.md)
are canonical.
## HTTP Artifact Security Boundary
Ordinary CLI file loading is provided by Promptkit. Scriptorium's HTTP adapter
injects a restricted `promptkit.ArtifactReader` for inbound HTTP requests.
That reader denies file references without an artifact root, enforces the
configured byte limit, and applies Scriptorium's lexical root-containment rule.
The operating system still follows symlinks after the lexical check.
The [HTTP API](../api.md) owns observable request outcomes, and
[operations](../operations.md) owns deployment permissions and root selection.
## State, Errors, And Secrets
Scriptorium has no durable run-state store, checkpoint, cache, or resume
mechanism. Recovery is a new request after correcting inputs, configuration, or
environment.
Adapters map Promptkit public error identities into CLI exits or HTTP statuses
without classifying by message text. Raw API keys are not accepted in
Scriptorium configuration, CLI arguments, or HTTP payloads, and resolved
secrets must not be emitted.
## Architectural Invariants
- External YAML and JSON decoding remains strict.
- CLI and HTTP behavior remains presentation and transport logic rather than
framework orchestration.
- Explicit numeric request overrides preserve presence, including zero.
- HTTP artifact containment and byte limits remain Scriptorium policy.
- No application package depends on Promptkit implementation packages.
## Non-Goals
- Do not recreate an in-process Scriptorium framework API or compatibility
facade.
- Do not copy Promptkit types, defaults, built-in profiles, or implementation
into Scriptorium.
- Do not move CLI, inbound HTTP, process, or deployment policy into Promptkit.
- Do not add durable workflow, archive, or resume behavior.
Work that is not implemented belongs in `docs/roadmap/`.

View File

@@ -0,0 +1,165 @@
# Documentation Policy
## Purpose
This policy assigns each documentation topic to one canonical owner. Its goal is
to keep this repository's documentation accurate, concise, discoverable, and
resistant to drift for users, operators, developers, integrators, and LLM
coding agents.
## Core Rules
### One Canonical Owner
Each authoritative fact belongs in one document. A non-owning document may give
a short, stable summary for orientation, but it must link to the canonical owner
instead of repeating volatile details.
Volatile details include commands, flags, configuration fields and defaults,
module keys, schemas, file names, paths, status codes, retry behavior, and
runtime guarantees. If readers could reasonably treat a statement as a
contract, maintain it only in the owning document.
Minimal tested usage examples are allowed outside the owning contract when this
policy assigns them an orientation or instructional purpose. They must link to
the canonical contract and must not redefine complete syntax, defaults, or
semantics.
### Current And Future Behavior
Outside `docs/roadmap/`, documentation describes implemented behavior only.
Partial features may be described only to their implemented boundary.
ADRs are the narrow exception: an ADR may record an accepted architectural
decision before implementation, but acceptance must not be presented as proof
that the behavior exists. The roadmap owns implementation status and sequencing
until the decision is implemented. Current architecture, user, operator,
integration, and internal documentation are updated when the behavior lands.
### Audience And Detail
Write for the document's stated audience and include only the detail needed for
its owned topic. User and operator docs should not expose implementation detail.
Developer docs should link to user-facing and external contracts rather than
restate them.
### Examples
Complete copyable files belong in `examples/`. Documentation may use the
smallest illustrative snippet needed to explain its owned topic, but should link
to maintained examples instead of embedding a second complete copy.
Examples must be valid, secret-free, and tested where practical. Commands and
configuration used in documentation should match the application.
### Security And Privacy
Documentation and examples must not contain real credentials, private keys,
private environment dumps, sensitive source material, or private infrastructure
details unless intentionally public. Document secret-handling mechanisms, not
secret values.
## Canonical Ownership
| Topic | Canonical owner | Owned content | Content owned elsewhere |
| --- | --- | --- | --- |
| Product orientation and minimal end-to-end quickstart | `README.md` | What this project is, why it is useful, one shortest successful invocation, and links onward. | Complete command reference, configuration reference, operational procedures, implementation detail. |
| Contributor entry point | `docs/development.md` | Task-oriented reading guide, minimal contributor orientation, baseline validation commands, and links to canonical docs. | Package inventory, architecture rules, subsystem behavior, and detailed change recipes, which belong in the relevant internal component document. |
| Current application architecture | `docs/policy/architecture.md` | System shape, normative ownership, dependency direction, architectural boundaries, invariants, safety properties, and non-goals. | Concrete package inventory, implementation mechanics, contributor procedures, decision history, future work. |
| Documentation organization | `docs/policy/documentation.md` | Documentation ownership, audience boundaries, maintenance rules, and ADR/document lifecycle. | Application architecture or product behavior. |
| Testing policy | `docs/policy/testing.md` | Test philosophy, risk-based sufficiency, test boundaries, doubles, coverage guidance, regression-test policy, and criteria for adding, rewriting, or deleting tests. | Subsystem behavior, application contracts, subsystem-specific test inventories, and implementation plans. |
| CLI contract | `docs/cli.md` | Commands, arguments, flags, invocation semantics, and exit codes. | End-to-end operating procedures, configuration field definitions, runtime filesystem layout, module implementation details. |
| Configuration contract | `docs/config.md` | Application discovery and precedence, source locations, server fields, render default, HTTP limits, and credential mapping. | Promptkit framework formats and defaults, complete example files, CLI syntax, runtime lifecycle, and implementation detail. |
| Operations | `docs/operations.md` | Runtime workflows, physical filesystem and state layout, output, cache, and debug handling, resume, cleanup, permissions, recovery, and operational limits. | CLI flag syntax, configuration field definitions, logical output schemas, implementation mechanics. |
| Release procedure | `docs/release.md` | Candidate validation, version and tag operations, hosted-workflow observation, and published-artifact verification. | Runtime operations, version-specific announcements, and complete application-interface contracts. |
| Version-specific release notes | `docs/releases/` | Immutable release summaries, compatibility notices, and migration announcements for one published version. | Complete CLI, HTTP, configuration, operations, or dependency contracts. |
| Public HTTP contract | `docs/api.md` | Routes, authentication, media types, request and response schemas, status codes, pagination, caching, idempotency, rate limits, and HTTP retry semantics. | Client walkthroughs, upstream or downstream integration internals, implementation detail. |
| Consumer guidance | `docs/consumers/` | Choosing between Scriptorium's executable interfaces and understanding consumer responsibilities. | HTTP wire semantics, CLI syntax, Promptkit's Go package, and internal implementation detail. |
| External and durable integration contracts | `docs/integrations/` | Scriptorium-owned process and executable integration contracts. | Promptkit framework formats and outbound provider protocols, physical runtime placement, internal transformations, CLI syntax, and configuration defaults. |
| Implemented component inventory | `docs/internal/overview.md` | Current packages and components, their implemented responsibilities, and links to focused internal docs. | Normative architecture, contributor reading policy, external contracts. |
| Internal component behavior | Other files under `docs/internal/` | Implementation flow, internal collaborators and state transitions, package-local guarantees and failures, and relevant tests. | Global architecture invariants, configuration definitions and defaults, external schemas, operator procedures. |
| Architectural decision history | `docs/adr/` | Significant decisions, context, alternatives, rationale, consequences, and supersession history. | Current behavior reference, implementation status, task sequencing. |
| Future work and implementation status | `docs/roadmap/` | Proposed, accepted, deferred, or rejected work; implementation status; sequencing; and task breakdowns. | Implemented behavior reference and architectural decision rationale. |
| Complete copyable artifacts | `examples/` | Maintained configuration, inputs, and other files intended to be copied or run. | Field-by-field reference, command reference, prose explanation. |
Documents that do not exist are required only when the corresponding interface
or responsibility exists. Do not create placeholder API, consumer, integration,
or operations documents for behavior the application does not have.
## Boundary Rules
### Orientation
The README owns product orientation. The developer guide routes contributors.
Architecture owns normative structure. Internal overview owns the current
concrete component map. These documents may link to one another but should not
maintain parallel package or behavior descriptions.
### Commands, Configuration, And Operations
CLI documentation answers how to invoke the application. Configuration
documentation answers what settings mean. Operations answers what happens to
runtime state and how to operate or recover the application. When a workflow
crosses these topics, choose the document that owns the task and link to the
other contracts.
### Contracts And Implementation
Integration and API documents define externally observable shapes and
semantics. Internal documents explain how this project implements or consumes
those contracts. Internal docs may name a field, file, or protocol to identify
a dependency, but must link to its canonical contract for the definition.
### Security Topics
This policy owns what documentation and examples may contain. Architecture owns
application security invariants. Configuration owns credential-supply
mechanisms. Operations owns permissions and handling of sensitive runtime
artifacts. Internal docs own implementation mechanisms only.
## Architecture Decision Records
Use sequentially numbered ADR filenames such as
`0001-record-architecture-decisions.md`. Follow the lightweight Nygard format:
1. title;
2. status;
3. date;
4. context;
5. decision;
6. alternatives considered;
7. consequences.
Use one of these statuses:
- **Proposed:** the decision is under consideration and may change;
- **Accepted:** the decision is approved, whether or not implementation is
complete;
- **Rejected:** the proposed decision was considered and not adopted;
- **Superseded:** a later ADR replaces the accepted decision.
A proposed ADR transitions to accepted or rejected. An accepted ADR transitions
to superseded only when a later accepted ADR replaces it. An ADR may be created
as accepted when the decision has already been made.
Treat the decision content of an accepted ADR as immutable. Its status and
supersession metadata may be updated, but a changed decision requires a new ADR.
A superseded ADR must link to its replacement, and the replacement must link
back to the superseded ADR. Rejected architectural alternatives belong in the
ADR; rejected product ideas belong in the roadmap.
## Maintenance
When behavior changes, update its canonical owner in the same change. If
ownership moves, remove the old definition and replace it with a link where
navigation remains useful.
Before completing documentation work:
- verify affected behavior and examples;
- check commands, flags, fields, defaults, schemas, and paths against their
implementation;
- keep unimplemented behavior in the roadmap, subject to the ADR exception;
- remove stale references and validate links;
- confirm that non-owning documents summarize and link rather than redefine;
- confirm that no secrets or sensitive private data were added.

301
docs/policy/testing.md Normal file
View File

@@ -0,0 +1,301 @@
# Testing Policy
## Purpose
Our tests exist to make **incorrect changes expensive and correct changes cheap**.
We do not optimize for test count, line coverage, exhaustive isolation, or the fewest possible tests. We optimize for sufficient confidence in important behavior while imposing as little unnecessary friction as possible on future development.
## Every test has a cost
Testing is not an unqualified good. Every test imposes both an immediate cost and a continuing lifetime cost.
A test must be:
- written and reviewed;
- understood by future maintainers and coding agents;
- executed in local and CI workflows;
- diagnosed when it fails;
- updated when legitimate behavior changes;
- maintained as fixtures, APIs, and dependencies evolve; and
- removed or rewritten when it becomes redundant, brittle, misleading, or obsolete.
Tests also create cognitive and architectural friction. They can constrain refactoring, duplicate policy, slow feedback loops, add noise to failures, and cause harmless implementation changes to require unrelated edits across the suite.
A test is warranted only when the confidence it provides justifies these costs.
Apply this cost-benefit analysis at two levels:
1. **Per test:** What realistic defect does this test detect, how consequential would that defect be, and is that protection worth the test's lifetime cost?
2. **Across the suite:** Does this collection provide materially more confidence than a smaller, simpler suite would?
The preferred test suite is a **lean suite that provides sufficient confidence in the risks that matter, without redundant or low-value tests**. We seek sufficient confidence with the least unnecessary testing friction, not the fewest possible tests.
Some friction is intentional. Tests should make dangerous changes—such as breaking compatibility, corrupting data, violating security boundaries, or reintroducing subtle bugs—require deliberate review. They should not make ordinary internal changes needlessly expensive.
The cost of a test is not a reason to omit testing by default. Do not cite maintenance cost abstractly. When omitting a plausible test, be able to state why the protected failure is low-risk, already covered, obvious, reversible, or cheaper to detect elsewhere. For consequential, subtle, or difficult-to-observe behavior, the presumption should favor testing.
## Default testing style
Use a **classical/Detroit-style** approach:
- Test observable behavior, resulting state, contracts, and invariants.
- Use real internal collaborators when they are fast and deterministic.
- Use fakes, stubs, or mocks primarily at expensive, nondeterministic, destructive, or external boundaries.
- Prefer package-level behavioral tests over tests coupled to private helpers or internal call sequences.
- Treat exact collaborator interactions as testable behavior only when the interaction itself is a requirement.
Examples of appropriate seams include clocks, randomness, subprocesses, remote APIs, object storage, email, and paid LLM calls.
## Test execution requirements
Tests in the default suite must be deterministic, offline, and independent of real credentials. They must not invoke paid APIs or depend on mutable external services. Tests that require live infrastructure must be explicitly opt-in and clearly separated from the default suite.
Control clocks, randomness, environment variables, and other process-global or machine-specific state when they affect behavior. Tests should be safe to run repeatedly and alongside other tests without depending on execution order or state left by an earlier test.
## What deserves tests
Prioritize tests for:
1. Public and package-level contracts.
2. Domain rules and important invariants.
3. Boundary conditions and malformed input.
4. Failure handling, cancellation, retries, recovery, and partial success.
5. Serialization, schemas, compatibility, and round trips.
6. Previously observed or plausible regressions.
7. Representative integration and end-to-end workflows.
A package-level contract is behavior relied upon by another package or major collaborator, not every observable detail of a package implementation.
For behavior involving **data integrity, destructive operations, compatibility, security, concurrency, idempotency, or recovery**, presume that durable tests are required unless the behavior is already credibly protected at another layer.
Do not add tests merely because a function, branch, or line exists. Do not add a test when the same meaningful risk is already adequately protected elsewhere.
## Choose the right test boundary
Test through the narrowest stable boundary that expresses the behavior clearly.
This is often the package API, but it may instead be:
- a smaller pure function when dense application logic is most clearly isolated there;
- a package-level operation when several internal collaborators jointly produce the behavior; or
- a larger integration boundary when correctness emerges from interaction with a real dependency.
Do not force all behavior through oversized end-to-end tests. Do not test every private helper merely because it exists. Choose the boundary that gives durable confidence with the least incidental coupling.
## Test behavior, not implementation
A test should protect a decision, contract, or invariant—not memorialize the current implementation.
Before adding or retaining a test, ask:
> What realistic defect would this test catch?
A test is suspect when its main purpose is to detect that someone:
- changed an internal constant;
- renamed or split a private helper;
- reordered equivalent internal operations;
- changed incidental formatting;
- replaced one correct algorithm with another; or
- refactored internal object structure without changing behavior.
Refactoring should normally require no test edits unless the refactored structure is itself part of the contract.
A test can be factually correct and still have negative value. Accurately describing current behavior is not enough; the protected behavior must be important enough to justify the future friction.
## Expected effects of different changes
Use the following expectations when evaluating test failures and test maintenance:
| Change | Expected effect on tests |
|---|---|
| Internal refactor that preserves behavior | Existing tests should normally remain unchanged and continue to pass. |
| Change to an internal default with no contractual significance | Behavioral tests should normally remain unchanged; tests should derive expectations from configuration or relationships rather than duplicate the old value. |
| Intentional change to public behavior, policy, schema, or compatibility guarantees | The relevant tests should be reviewed and changed deliberately. |
| Accidental violation of a contract or invariant | Tests should fail; fix the production code rather than rewriting the tests to accept the defect. |
A test failing is not the same as a test needing to be edited. Many tests may correctly fail because of one production defect. The maintenance smell is a correct internal change that requires unrelated expectation updates throughout the suite.
## Separate mechanism from policy
Configurable thresholds and defaults must not be duplicated throughout the test suite.
For example, do not encode an internal concurrency limit indirectly:
```go
// Production policy:
const maxConcurrency = 4
// Brittle test:
err := startProcesses(5)
require.Error(t, err)
```
Instead, test the mechanism relationally:
```go
const limit = 2
runner := NewRunner(limit)
require.NoError(t, runner.Start(limit))
require.ErrorIs(t, runner.Start(limit+1), ErrTooMuchConcurrency)
```
The test should prove:
- the configured limit is accepted; and
- one beyond the configured limit is rejected.
The production default should be tested exactly only when its literal value is itself a public, operational, safety, protocol, or compatibility requirement.
Apply the same rule to limits, timeouts, capacities, retry counts, and ranges: test relationships and behavior, not duplicated literals.
For concurrency limits, test both kinds of behavior when relevant:
1. **Configuration enforcement:** invalid or excessive requested values are handled correctly.
2. **Runtime enforcement:** observed peak concurrency never exceeds the configured limit.
Use a test-controlled limit and measure the behavior relative to that limit. Do not merely assert today's default value.
## Avoid semantic duplication across layers
Each behavior should have a clear test owner.
- Configuration tests own application YAML, discovery, precedence, and
application defaults.
- CLI tests own argument mapping, streams, summaries, exit behavior, and
representative command workflows.
- HTTP tests own DTOs, strict decoding, limits, status mapping, and restricted
artifact policy.
- Formatter tests own prepared-run text and JSON presentation.
- Architecture tests own dependency direction and removal invariants.
- Promptkit owns framework parsing, orchestration, validation, profiles, and
model-client behavior.
Higher-level tests should not repeat every lower-level case. A single intentional policy change should not require unrelated edits across many test files.
Tests that are individually reasonable may still be collectively redundant. Evaluate the marginal value of each additional test in light of the protection already provided by the rest of the suite.
## Use test doubles deliberately
Choose the least elaborate test double that provides the required control or observation.
As a default:
1. Prefer real collaborators when they are fast and deterministic.
2. Use small in-memory fakes when realistic stateful behavior is helpful.
3. Use stubs when a dependency only needs to provide controlled responses.
4. Use mocks when the interaction itself is contractual.
Mocks are appropriate when the contract includes facts such as:
- a notification is sent exactly once;
- a transaction is committed only after successful writes;
- cancellation reaches a subprocess;
- an expensive API is called no more than once; or
- a security audit event is emitted.
Do not use mocks merely to isolate every object or reproduce the implementation's call graph.
## Go-specific guidance
Use:
- table-driven tests for meaningful behavioral categories and boundaries;
- `t.TempDir()` for real filesystem behavior;
- `httptest.Server` for realistic HTTP interactions;
- fuzz tests for parsers, normalization, path handling, and broad input spaces;
- golden files only when the complete output is intentionally stable;
- integration tests where correctness depends on component interaction; and
- a small number of representative end-to-end tests.
Avoid exact error-string assertions unless the wording is itself contractual. Prefer `errors.Is`, `errors.As`, typed errors, or structured error fields.
At CLI boundaries, prefer exit classifications, structured output, and the smallest stable semantic fragment needed to identify the error. Do not snapshot complete diagnostic wording unless it is contractual.
Golden-file updates must require an explicit local flag. CI must not update golden files automatically, and reviewers must inspect the semantic diff before accepting an update.
Keep tests readable and direct. Test helpers and fixture frameworks must earn their own maintenance cost; do not build elaborate test infrastructure for small or isolated needs.
## Coverage
Coverage is a diagnostic, not a target.
Use it to find untested critical branches and unexpectedly weak packages. Do not write low-value tests solely to increase a percentage, and do not infer test quality from coverage alone.
Security-sensitive HTTP containment and external mappings may warrant denser
coverage than straightforward process wiring. Uneven coverage is acceptable
when it reflects risk.
Increasing coverage is valuable only when the newly covered behavior protects a meaningful risk at an acceptable cost.
## Regression tests
A bug fix should normally include a regression test that fails before the fix and passes afterward.
Retain the test when the defect could realistically recur and its consequences justify the ongoing cost. Prefer the narrowest durable test of the violated contract or invariant; do not preserve accidental implementation details from the original bug.
Not every historical bug requires a permanent test. If the underlying design has made recurrence impossible, the test has become redundant, or a stronger invariant test now subsumes it, remove or consolidate it.
## Deleting or rewriting tests
Tests are maintained code, not permanent historical artifacts.
Delete or rewrite a test when its maintenance cost exceeds the confidence it provides.
Strong candidates include tests that:
- require updates after harmless internal changes;
- directly assert private constants without protecting a real contract;
- duplicate the same policy across several layers;
- verify mock choreography rather than outcomes;
- snapshot large amounts of incidental output;
- test trivial private helpers already exercised through stable package behavior;
- protect risks already covered more effectively elsewhere;
- are flaky, misleading, obsolete, or disproportionately expensive to diagnose; or
- no longer correspond to a plausible failure mode.
Several brittle tests may encode one genuine requirement. Replace them with one durable behavior-level or invariant test rather than preserving all of them.
Deleting a low-value test can improve the quality of the suite by reducing noise, maintenance burden, and friction around legitimate change.
## Reviewing a proposed test
Use the following questions when the value, boundary, or durability of a proposed test is not self-evident. Significant test additions should be reviewable against them, but written answers are not required for every routine test.
1. What realistic defect would it catch?
2. How likely is that defect?
3. How consequential would it be?
4. Is the behavior already protected elsewhere?
5. At which layer should this behavior be owned?
6. Does the test assert a durable contract or an incidental implementation detail?
7. Could the implementation be refactored without changing the behavior and without editing this test?
8. What should cause this test to fail?
9. What legitimate changes should not cause this test to fail?
10. What ongoing maintenance, execution, and diagnostic cost will the test impose?
11. Is there a smaller or more direct test that protects the same risk?
Do not add the test when its expected lifetime cost exceeds its expected protective value.
When deciding not to test plausible behavior, record or be able to explain why the risk is low, already protected, obvious, reversible, or cheaper to detect elsewhere.
## Definition of sufficient
A test suite is sufficient when:
- important contracts and invariants are protected;
- meaningful boundaries and failure modes are exercised;
- realistic and consequential regressions are credibly protected against silent recurrence;
- behavior involving data integrity, destructive operations, compatibility, security, concurrency, idempotency, and recovery is credibly protected;
- important external boundaries have realistic integration coverage;
- representative complete workflows are tested;
- failures provide useful signal rather than redundant noise;
- legitimate internal changes usually do not require test edits; and
- additional tests would mostly repeat existing protection or preserve inconsequential implementation details.
Sufficiency is a risk judgment, not a coverage percentage or test count. Reassess it as the application, its users, and the consequences of failure evolve.
The governing rule is:
> Test heavily where failure is consequential, subtle, or difficult to detect after the fact. Test lightly where failure is obvious, reversible, and inexpensive—and retain no test whose lifetime cost exceeds the confidence it provides.

376
docs/release.md Normal file
View File

@@ -0,0 +1,376 @@
# Release Procedure
## Release Model And Status
Scriptorium publishes annotated semantic tags and tag-triggered Linux binary
releases. The hosted
[release workflow](../.woodpecker/release.yml) builds `amd64` and `arm64`
executables, publishes their SHA-256 checksums, and uses the matching file
under `docs/releases/` as the hosted release body.
`v0.12.0` is the first published application-only release. For each later
release, select a new `vMAJOR.MINOR.PATCH` version according to the intended
compatibility change. A selected version remains an unreleased candidate until
its annotated tag is published, the hosted workflow succeeds, and every
published artifact is verified.
Run this procedure from the Scriptorium repository root. A release must not
depend on a Go workspace, module replacement, vendor tree, sibling checkout,
unpublished dependency, or unpushed source commit.
## Establish The Candidate
Start a POSIX shell, choose a semantic version that has not been published, and
export it as `RELEASE_VERSION`. For example, if `v0.12.1` is the intended next
version and remains unpublished, select:
```sh
export RELEASE_VERSION=v0.12.1
```
Use the version appropriate to the actual compatibility change rather than
assuming that the example is the next release. Then run the following guard in
that same shell:
```sh
set -eu
: "${RELEASE_VERSION:?export an unpublished vMAJOR.MINOR.PATCH version}"
if ! printf '%s\n' "$RELEASE_VERSION" |
grep -Eq '^v(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)\.(0|[1-9][0-9]*)$'
then
printf '%s\n' "invalid release version: $RELEASE_VERSION" >&2
exit 1
fi
RELEASE_COMMIT=$(git rev-parse --verify 'HEAD^{commit}')
export RELEASE_COMMIT
check_release_candidate() {
test "$(git branch --show-current)" = main
test -z "$(git status --porcelain)"
gowork_value=$(go env GOWORK)
case "$gowork_value" in
''|off) ;;
*)
printf '%s\n' "active Go workspace: $gowork_value" >&2
return 1
;;
esac
test -z "$(git ls-files go.work go.work.sum)"
test ! -e vendor
if grep -Eq '^[[:space:]]*replace([[:space:]]|\()' go.mod
then
printf '%s\n' 'go.mod contains a replacement' >&2
return 1
fi
git fetch origin main --tags
test "$RELEASE_COMMIT" = \
"$(git rev-parse --verify 'refs/remotes/origin/main^{commit}')"
if git show-ref --verify --quiet "refs/tags/$RELEASE_VERSION"
then
printf '%s\n' "local tag already exists: $RELEASE_VERSION" >&2
return 1
fi
if test -n "$(
git ls-remote --tags origin \
"refs/tags/$RELEASE_VERSION" \
"refs/tags/$RELEASE_VERSION^{}"
)"
then
printf '%s\n' "remote tag already exists: $RELEASE_VERSION" >&2
return 1
fi
}
check_release_candidate
```
Do not continue unless this guard succeeds. It deliberately requires the
candidate to be the exact clean commit already published at `origin/main`.
## Verify Modules And Repository Boundaries
Confirm the module path and declared Go version:
```sh
test "$(
GOWORK=off go list -m -f '{{.Path}} {{.GoVersion}}'
)" = 'gitea.maximumdirect.net/eric/scriptorium 1.25.5'
```
Require Promptkit `v0.1.0` as both the direct module-graph edge and the selected
module version:
```sh
direct_promptkit=$(
GOWORK=off go mod graph |
awk '
$1 == "gitea.maximumdirect.net/eric/scriptorium" &&
$2 ~ /^gitea\.maximumdirect\.net\/eric\/promptkit@/ {
print $2
}
'
)
test "$direct_promptkit" = \
'gitea.maximumdirect.net/eric/promptkit@v0.1.0'
test "$(
GOWORK=off go list -m -f '{{.Path}}@{{.Version}}' \
gitea.maximumdirect.net/eric/promptkit
)" = 'gitea.maximumdirect.net/eric/promptkit@v0.1.0'
GOWORK=off go list -m all
```
Require tidy module metadata and recheck the repository exclusions:
```sh
GOWORK=off go mod tidy -diff
test -z "$(git ls-files go.work go.work.sum)"
test ! -e vendor
if grep -Eq '^[[:space:]]*replace([[:space:]]|\()' go.mod
then
printf '%s\n' 'go.mod contains a replacement' >&2
exit 1
fi
test -z "$(git status --porcelain)"
```
## Validate The Application
Run the complete application validation:
```sh
GOWORK=off go test ./...
GOWORK=off go test -race ./...
GOWORK=off go vet ./...
validation_build_dir=$(mktemp -d)
GOWORK=off go build \
-o "$validation_build_dir/scriptorium" \
./cmd/scriptorium
```
The ordinary test run includes the architecture guard that rejects a root Go
package, former framework package families, and imports of Promptkit internal
packages. Inspect the repository for generated binaries, credentials,
temporary output, sibling paths, and other files that do not belong in the
tracked release source.
Check every tracked Go file. This command must produce no output:
```sh
unformatted=$(
git ls-files '*.go' |
while IFS= read -r go_file
do
gofmt -l "$go_file"
done
)
test -z "$unformatted"
```
Run the maintained render script and smoke-test both maintained configuration
examples without a model call:
```sh
GOWORK=off ./examples/render-markdown-summary.sh
for config_file in examples/config.yml examples/config.full.yml
do
GOWORK=off go run ./cmd/scriptorium render \
--config "$config_file" \
--prompt generic.markdown_summary \
--input transcript=./examples/fixtures/transcript.md \
--input glossary=./examples/fixtures/glossary.yml \
--format json >/dev/null
done
```
Exercise usage output and offline rendering with the temporary native
executable:
```sh
usage_output="$validation_build_dir/usage.txt"
if "$validation_build_dir/scriptorium" >"$usage_output" 2>&1
then
printf '%s\n' 'expected an invocation without a command to fail' >&2
exit 1
fi
grep -F 'usage: scriptorium' "$usage_output"
"$validation_build_dir/scriptorium" render \
--config ./examples/config.yml \
--prompt generic.markdown_summary \
--input transcript=./examples/fixtures/transcript.md \
--input glossary=./examples/fixtures/glossary.yml \
--format text >/dev/null
```
Follow every maintained local, Promptkit-tagged, and other external Markdown
link. Confirm that all repository-relative link targets exist. Finish the
application checks with:
```sh
git diff --check
test -z "$(git status --porcelain)"
```
## Require Release Notes And Reproduce Packaging
The immutable release note must exist before tagging:
```sh
release_notes="docs/releases/$RELEASE_VERSION.md"
test -f "$release_notes"
test -s "$release_notes"
```
Validate every local and currently published link in the note. For links
pinned to the candidate Scriptorium tag, confirm that the corresponding
repository-relative path exists even though its tag URL is not live yet.
Reproduce the hosted build flags, targets, and filenames in a temporary
directory:
```sh
release_dist=$(mktemp -d)
release_package='gitea.maximumdirect.net/eric/scriptorium/cmd/scriptorium'
build_release_binary() {
target_os="$1"
target_arch="$2"
output="$release_dist/scriptorium-$RELEASE_VERSION-$target_os-$target_arch"
CGO_ENABLED=0 GOOS="$target_os" GOARCH="$target_arch" GOWORK=off \
go build -trimpath -ldflags '-s -w' \
-o "$output" "$release_package"
}
build_release_binary linux amd64
build_release_binary linux arm64
test -s "$release_dist/scriptorium-$RELEASE_VERSION-linux-amd64"
test -s "$release_dist/scriptorium-$RELEASE_VERSION-linux-arm64"
file "$release_dist/scriptorium-$RELEASE_VERSION-linux-amd64"
file "$release_dist/scriptorium-$RELEASE_VERSION-linux-arm64"
```
Require `file` to identify Linux executables for `x86-64` and `ARM aarch64`,
respectively. Inspect the
[hosted workflow](../.woodpecker/release.yml) and confirm that it uses the
same build flags and names, copies the selected release note to
`dist/RELEASE_NOTES.md`, publishes only `dist/scriptorium-*`, and keeps
checksum generation enabled.
## Create And Publish The Tag
Run the candidate guard again immediately before creating the tag:
```sh
check_release_candidate
test -f "$release_notes"
test -s "$release_notes"
```
Create an annotated tag explicitly bound to the validated commit, using the
version-specific release note as its message:
```sh
git tag --annotate "$RELEASE_VERSION" \
--file "$release_notes" \
"$RELEASE_COMMIT"
```
Inspect the tag and require it to resolve to the validated source:
```sh
test "$(git cat-file -t "refs/tags/$RELEASE_VERSION")" = tag
git show --no-patch --decorate "refs/tags/$RELEASE_VERSION"
test "$(
git rev-parse --verify "refs/tags/$RELEASE_VERSION^{commit}"
)" = "$RELEASE_COMMIT"
```
If inspection finds an error, delete only the unpublished local tag, correct
the candidate, and repeat the complete validation. Never move or recreate a
published tag.
Push only the selected tag ref:
```sh
git push origin \
"refs/tags/$RELEASE_VERSION:refs/tags/$RELEASE_VERSION"
```
## Observe And Verify Publication
Open the hosted workflow run for the selected tag. Require
`build-release-assets` to succeed before `publish-release`, then require the
publication step and hosted release to succeed. A queued, running, failed, or
partially published workflow is not a verified release.
Compare the local and remote annotated-tag objects and their source commits:
```sh
remote_tag=$(
git ls-remote --tags origin "refs/tags/$RELEASE_VERSION" |
awk 'NR == 1 { print $1 }'
)
remote_commit=$(
git ls-remote --tags origin "refs/tags/$RELEASE_VERSION^{}" |
awk 'NR == 1 { print $1 }'
)
test -n "$remote_tag"
test "$remote_tag" = \
"$(git rev-parse --verify "refs/tags/$RELEASE_VERSION")"
test "$remote_commit" = "$RELEASE_COMMIT"
```
Download the hosted binaries and checksum file into a temporary directory:
```sh
release_base="https://gitea.maximumdirect.net/eric/scriptorium/releases/download/$RELEASE_VERSION"
download_dir=$(mktemp -d)
(
cd "$download_dir"
for asset in \
"scriptorium-$RELEASE_VERSION-linux-amd64" \
"scriptorium-$RELEASE_VERSION-linux-arm64" \
SHA256SUMS
do
curl --fail --location --remote-name "$release_base/$asset"
done
test -s "scriptorium-$RELEASE_VERSION-linux-amd64"
test -s "scriptorium-$RELEASE_VERSION-linux-arm64"
test -s SHA256SUMS
sha256sum --check SHA256SUMS
test "$(wc -l < SHA256SUMS | tr -d ' ')" = 2
file "scriptorium-$RELEASE_VERSION-linux-amd64"
file "scriptorium-$RELEASE_VERSION-linux-arm64"
)
```
Require the same Linux architectures observed in the local packaging check and
confirm that the hosted release contains no unexpected asset. On a compatible
Linux host, make the matching downloaded binary executable and repeat the
usage-output and offline-render smoke checks against it.
Only after the tag, workflow, release body, binaries, architectures, and
checksums all pass verification is the candidate a verified published release.
## Handle Failures
Before tag publication, correct the release commit or note and restart the
complete procedure. After tag publication, never delete, move, overwrite, or
recreate the tag. A transient hosted failure may be retried only against the
same immutable tag and commit and only when doing so cannot overwrite or
silently retain partial assets. A source, packaging, note, or artifact defect
requires a new corrective semantic version from a new validated commit.
Record the selected version, validated commit, tag object, workflow result,
artifact names, checksum result, and smoke-check outcome in the release
checkpoint. Keep temporary builds and downloaded assets outside the repository,
and require a clean `main` synchronized with `origin/main` when verification
is complete.

36
docs/releases/v0.12.0.md Normal file
View File

@@ -0,0 +1,36 @@
# Scriptorium v0.12.0
## Breaking Project Boundary
Scriptorium is now an executable-only CLI and HTTP application. This is a
breaking change for Go consumers: the former root Go package is not included,
and no compatibility facade is provided.
Scriptorium `v0.11.1` was the final framework-bearing release. Former Go
consumers should follow the
[migration guide](https://gitea.maximumdirect.net/eric/scriptorium/src/tag/v0.12.0/docs/consumers/migrating-to-promptkit.md)
and adopt
[Promptkit `v0.1.0`](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/consumers/pkg-promptkit.md)
for in-process prompt preparation and execution.
## Application Interfaces
The Scriptorium command-line and HTTP application interfaces remain. Their
canonical documentation defines the supported commands, configuration,
requests, responses, operational behavior, and deployment responsibilities:
- [CLI reference](https://gitea.maximumdirect.net/eric/scriptorium/src/tag/v0.12.0/docs/cli.md)
- [HTTP API reference](https://gitea.maximumdirect.net/eric/scriptorium/src/tag/v0.12.0/docs/api.md)
- [Configuration reference](https://gitea.maximumdirect.net/eric/scriptorium/src/tag/v0.12.0/docs/config.md)
- [Operations guide](https://gitea.maximumdirect.net/eric/scriptorium/src/tag/v0.12.0/docs/operations.md)
## Framework Dependency And Consumers
The released Scriptorium binaries use Promptkit `v0.1.0` as their framework
dependency. Promptkit owns the reusable engine, source formats, profiles,
generation boundary, and validation contracts. See the
[Promptkit Go consumer guide](https://gitea.maximumdirect.net/eric/promptkit/src/tag/v0.1.0/docs/consumers/pkg-promptkit.md)
for that supported API.
All known downstream Go consumers were migrated to Promptkit before this
release.

13
examples/config.full.yml Normal file
View File

@@ -0,0 +1,13 @@
prompt_dir: ./examples/prompts
profile_dir: ./examples/profiles
schema_dir: ./examples/schemas
server:
addr: 127.0.0.1:8080
artifact_root: .
max_request_bytes: 16777216
max_artifact_bytes: 16777216
max_response_bytes: 16777216
defaults:
render_format: text

View File

@@ -1,9 +1,10 @@
prompt_dir: ./prompts prompt_dir: ./examples/prompts
profile_dir: ./profiles profile_dir: ./examples/profiles
schema_dir: ./schemas schema_dir: ./examples/schemas
server: server:
addr: :8080 addr: :8080
artifact_root: .
defaults: defaults:
render_format: text render_format: text

18
examples/http-run.json Normal file
View File

@@ -0,0 +1,18 @@
{
"prompt_id": "generic.markdown_summary",
"profile_id": "local-fast",
"inputs": {
"transcript": {
"type": "file",
"uri": "./examples/fixtures/transcript.md"
},
"glossary": {
"type": "file",
"uri": "./examples/fixtures/glossary.yml"
}
},
"vars": {
"session_date": "2026-05-04"
},
"include_raw_output": false
}

View File

@@ -0,0 +1,13 @@
#!/usr/bin/env bash
set -euo pipefail
repo_root="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
cd "$repo_root"
go run ./cmd/scriptorium render \
--config ./examples/config.yml \
--prompt generic.markdown_summary \
--input transcript=./examples/fixtures/transcript.md \
--input glossary=./examples/fixtures/glossary.yml \
--format text

7
go.mod
View File

@@ -3,8 +3,11 @@ module gitea.maximumdirect.net/eric/scriptorium
go 1.25.5 go 1.25.5
require ( require (
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2 gitea.maximumdirect.net/eric/promptkit v0.1.0
gopkg.in/yaml.v3 v3.0.1 gopkg.in/yaml.v3 v3.0.1
) )
require golang.org/x/text v0.14.0 // indirect require (
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2 // indirect
golang.org/x/text v0.14.0 // indirect
)

2
go.sum
View File

@@ -1,3 +1,5 @@
gitea.maximumdirect.net/eric/promptkit v0.1.0 h1:vuKeBxkiY8E54LRFbLQFjlJJCiOfMvB1++DYBCrD/ug=
gitea.maximumdirect.net/eric/promptkit v0.1.0/go.mod h1:R95NM6fbMDGDC0/UomgnSBP6ui2ns+8SZb8bESNvrDQ=
github.com/dlclark/regexp2 v1.11.0 h1:G/nrcoOa7ZXlpoa/91N3X7mM3r8eIlMBBJZvsz/mxKI= github.com/dlclark/regexp2 v1.11.0 h1:G/nrcoOa7ZXlpoa/91N3X7mM3r8eIlMBBJZvsz/mxKI=
github.com/dlclark/regexp2 v1.11.0/go.mod h1:DHkYz0B9wPfa6wondMfaivmHpzrQ3v9q8cnmRbL6yW8= github.com/dlclark/regexp2 v1.11.0/go.mod h1:DHkYz0B9wPfa6wondMfaivmHpzrQ3v9q8cnmRbL6yW8=
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2 h1:KRzFb2m7YtdldCEkzs6KqmJw4nqEVZGK7IN2kJkjTuQ= github.com/santhosh-tekuri/jsonschema/v6 v6.0.2 h1:KRzFb2m7YtdldCEkzs6KqmJw4nqEVZGK7IN2kJkjTuQ=

View File

@@ -12,18 +12,11 @@ import (
"strings" "strings"
"time" "time"
"gitea.maximumdirect.net/eric/promptkit"
httpadapter "gitea.maximumdirect.net/eric/scriptorium/internal/adapter/http" httpadapter "gitea.maximumdirect.net/eric/scriptorium/internal/adapter/http"
artifactadapter "gitea.maximumdirect.net/eric/scriptorium/internal/artifact"
appconfig "gitea.maximumdirect.net/eric/scriptorium/internal/config" appconfig "gitea.maximumdirect.net/eric/scriptorium/internal/config"
"gitea.maximumdirect.net/eric/scriptorium/internal/defaults" "gitea.maximumdirect.net/eric/scriptorium/internal/defaults"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
renderformat "gitea.maximumdirect.net/eric/scriptorium/internal/format" renderformat "gitea.maximumdirect.net/eric/scriptorium/internal/format"
"gitea.maximumdirect.net/eric/scriptorium/internal/llm"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile"
"gitea.maximumdirect.net/eric/scriptorium/internal/prompt"
"gitea.maximumdirect.net/eric/scriptorium/internal/promptdef"
"gitea.maximumdirect.net/eric/scriptorium/internal/usecase"
"gitea.maximumdirect.net/eric/scriptorium/internal/validate"
) )
const ( const (
@@ -34,7 +27,6 @@ const (
const ( const (
errPromptDirRequired = "prompt directory is required; provide --prompt-dir or config.yml prompt_dir" errPromptDirRequired = "prompt directory is required; provide --prompt-dir or config.yml prompt_dir"
errProfileDirRequired = "profile directory is required; provide --profile-dir or config.yml profile_dir"
) )
type runConfig struct { type runConfig struct {
@@ -79,6 +71,22 @@ type serveConfig struct {
promptDir string promptDir string
profileDir string profileDir string
schemaDir string schemaDir string
artifactRoot string
maxRequestBytes int64
maxArtifactBytes int64
maxResponseBytes int64
}
type commonCommandSettings struct {
promptDir string
profileDir string
schemaDir string
serverAddr string
artifactRoot string
maxRequestBytes int64
maxArtifactBytes int64
maxResponseBytes int64
defaultRenderFormat renderformat.PreparedRunOutputFormat
} }
type listFlag []string type listFlag []string
@@ -125,24 +133,13 @@ func runCommand(args []string, stdout, stderr io.Writer) int {
return ExitRuntimeError return ExitRuntimeError
} }
llmClient, err := llm.NewOpenAICompatibleClient(llm.OpenAICompatibleConfig{ engine, err := newEngine(cfg)
Timeout: defaults.LLMRequestTimeoutDefault,
})
if err != nil { if err != nil {
fmt.Fprintf(stderr, "llm client error: %v\n", err) fmt.Fprintf(stderr, "engine error: %v\n", err)
return ExitRuntimeError return ExitRuntimeError
} }
runner := usecase.NewRunner( res, runErr := engine.Run(context.Background(), req)
promptdef.NewFilesystemRepository(cfg.promptDir),
profile.NewFilesystemRepository(cfg.profileDir),
artifactadapter.NewCompositeReader(),
prompt.NewGoRenderer(),
llmClient,
validate.NewStandardValidator(cfg.schemaDir),
)
res, runErr := runner.Run(context.Background(), req)
if runErr != nil { if runErr != nil {
fmt.Fprintf(stderr, "run error: %v\n", runErr) fmt.Fprintf(stderr, "run error: %v\n", runErr)
return ExitRuntimeError return ExitRuntimeError
@@ -170,16 +167,13 @@ func renderCommand(args []string, stdout, stderr io.Writer) int {
return ExitRuntimeError return ExitRuntimeError
} }
runner := usecase.NewRunner( engine, err := newEngine(&cfg.runConfig)
promptdef.NewFilesystemRepository(cfg.promptDir), if err != nil {
profile.NewFilesystemRepository(cfg.profileDir), fmt.Fprintf(stderr, "engine error: %v\n", err)
artifactadapter.NewCompositeReader(), return ExitRuntimeError
prompt.NewGoRenderer(), }
nil,
validate.NewStandardValidator(cfg.schemaDir),
)
prepared, prepErr := runner.Prepare(context.Background(), req) prepared, prepErr := engine.Prepare(context.Background(), req)
if prepErr != nil { if prepErr != nil {
fmt.Fprintf(stderr, "render error: %v\n", prepErr) fmt.Fprintf(stderr, "render error: %v\n", prepErr)
return ExitRuntimeError return ExitRuntimeError
@@ -205,24 +199,26 @@ func serveCommand(args []string, stderr io.Writer) int {
return ExitRuntimeError return ExitRuntimeError
} }
llmClient, err := llm.NewOpenAICompatibleClient(llm.OpenAICompatibleConfig{ artifactReader, err := httpadapter.NewRestrictedArtifactReader(cfg.artifactRoot, cfg.maxArtifactBytes)
Timeout: defaults.LLMRequestTimeoutDefault,
})
if err != nil { if err != nil {
fmt.Fprintf(stderr, "llm client error: %v\n", err) fmt.Fprintf(stderr, "artifact root error: %v\n", err)
return ExitRuntimeError return ExitRuntimeError
} }
runner := usecase.NewRunner( engine, err := newEngine(&runConfig{
promptdef.NewFilesystemRepository(cfg.promptDir), promptDir: cfg.promptDir,
profile.NewFilesystemRepository(cfg.profileDir), profileDir: cfg.profileDir,
artifactadapter.NewCompositeReader(), schemaDir: cfg.schemaDir,
prompt.NewGoRenderer(), }, promptkit.WithArtifactReader(artifactReader))
llmClient, if err != nil {
validate.NewStandardValidator(cfg.schemaDir), fmt.Fprintf(stderr, "engine error: %v\n", err)
) return ExitRuntimeError
}
h := httpadapter.NewHandler(runner) h := httpadapter.NewHandlerWithOptions(engine, httpadapter.HandlerOptions{
MaxRequestBytes: cfg.maxRequestBytes,
MaxResponseBytes: cfg.maxResponseBytes,
})
srv := &http.Server{ srv := &http.Server{
Addr: cfg.addr, Addr: cfg.addr,
Handler: h, Handler: h,
@@ -299,6 +295,10 @@ func parseServeArgs(args []string) (*serveConfig, error) {
fs.StringVar(&cfg.promptDir, "prompt-dir", "", "directory containing prompt definition YAML files") fs.StringVar(&cfg.promptDir, "prompt-dir", "", "directory containing prompt definition YAML files")
fs.StringVar(&cfg.profileDir, "profile-dir", "", "directory containing execution profile YAML files") fs.StringVar(&cfg.profileDir, "profile-dir", "", "directory containing execution profile YAML files")
fs.StringVar(&cfg.schemaDir, "schema-dir", "", "base directory for validation schemas") fs.StringVar(&cfg.schemaDir, "schema-dir", "", "base directory for validation schemas")
fs.StringVar(&cfg.artifactRoot, "artifact-root", "", "base directory for HTTP file input artifacts")
fs.Int64Var(&cfg.maxRequestBytes, "max-request-bytes", 0, "maximum HTTP request body bytes; 0 disables the limit")
fs.Int64Var(&cfg.maxArtifactBytes, "max-artifact-bytes", 0, "maximum HTTP file artifact bytes; 0 disables the limit")
fs.Int64Var(&cfg.maxResponseBytes, "max-response-bytes", 0, "maximum HTTP response body bytes; 0 disables the limit")
if err := fs.Parse(args); err != nil { if err := fs.Parse(args); err != nil {
return nil, err return nil, err
@@ -307,31 +307,41 @@ func parseServeArgs(args []string) (*serveConfig, error) {
return nil, fmt.Errorf("unexpected positional args: %v", fs.Args()) return nil, fmt.Errorf("unexpected positional args: %v", fs.Args())
} }
settings, err := resolveAppSettings(fs, cfg.configPath, appconfig.CLIOverrides{ settings, err := resolveCommonSettings(fs, cfg.configPath, appconfig.CLIOverrides{
PromptDir: cfg.promptDirIfSet(fs), PromptDir: cfg.promptDirIfSet(fs),
ProfileDir: cfg.profileDirIfSet(fs), ProfileDir: cfg.profileDirIfSet(fs),
SchemaDir: cfg.schemaDirIfSet(fs), SchemaDir: cfg.schemaDirIfSet(fs),
ServerAddr: cfg.addrIfSet(fs), ServerAddr: cfg.addrIfSet(fs),
ArtifactRoot: cfg.artifactRootIfSet(fs),
MaxRequestBytes: cfg.maxRequestBytesIfSet(fs),
MaxArtifactBytes: cfg.maxArtifactBytesIfSet(fs),
MaxResponseBytes: cfg.maxResponseBytesIfSet(fs),
}) })
if err != nil { if err != nil {
return nil, err return nil, err
} }
cfg.promptDir = settings.PromptDir cfg.promptDir = settings.promptDir
cfg.profileDir = settings.ProfileDir cfg.profileDir = settings.profileDir
cfg.schemaDir = settings.SchemaDir cfg.schemaDir = settings.schemaDir
cfg.addr = settings.ServerAddr cfg.addr = settings.serverAddr
cfg.artifactRoot = settings.artifactRoot
cfg.maxRequestBytes = settings.maxRequestBytes
cfg.maxArtifactBytes = settings.maxArtifactBytes
cfg.maxResponseBytes = settings.maxResponseBytes
if strings.TrimSpace(cfg.promptDir) == "" { if err := validateRequiredLibraryDirs(cfg.promptDir); err != nil {
return nil, errors.New(errPromptDirRequired) return nil, err
}
if strings.TrimSpace(cfg.profileDir) == "" {
return nil, errors.New(errProfileDirRequired)
} }
cfg.promptDir = filepath.Clean(cfg.promptDir) cfg.promptDir = filepath.Clean(cfg.promptDir)
if strings.TrimSpace(cfg.profileDir) != "" {
cfg.profileDir = filepath.Clean(cfg.profileDir) cfg.profileDir = filepath.Clean(cfg.profileDir)
}
cfg.schemaDir = filepath.Clean(cfg.schemaDir) cfg.schemaDir = filepath.Clean(cfg.schemaDir)
if strings.TrimSpace(cfg.artifactRoot) != "" {
cfg.artifactRoot = filepath.Clean(cfg.artifactRoot)
}
return cfg, nil return cfg, nil
} }
@@ -349,7 +359,7 @@ func registerExecutionRequestFlags(fs *flag.FlagSet, cfg *runConfig) {
fs.Float64Var(&cfg.temperature, "temperature", 0, "optional temperature override") fs.Float64Var(&cfg.temperature, "temperature", 0, "optional temperature override")
fs.IntVar(&cfg.maxTokens, "max-tokens", 0, "optional max tokens override") fs.IntVar(&cfg.maxTokens, "max-tokens", 0, "optional max tokens override")
fs.Float64Var(&cfg.topP, "top-p", 0, "optional top_p override") fs.Float64Var(&cfg.topP, "top-p", 0, "optional top_p override")
fs.DurationVar(&cfg.timeout, "timeout", defaults.LLMRequestTimeoutDefault, "LLM request timeout") fs.DurationVar(&cfg.timeout, "timeout", 0, "LLM request timeout")
fs.StringVar(&cfg.promptID, "prompt-id", "", "deprecated alias for --prompt") fs.StringVar(&cfg.promptID, "prompt-id", "", "deprecated alias for --prompt")
fs.StringVar(&cfg.profileID, "profile-id", "", "deprecated alias for --profile") fs.StringVar(&cfg.profileID, "profile-id", "", "deprecated alias for --profile")
} }
@@ -359,7 +369,7 @@ func finalizeExecutionRequestConfig(fs *flag.FlagSet, cfg *runConfig) error {
return fmt.Errorf("unexpected positional args: %v", fs.Args()) return fmt.Errorf("unexpected positional args: %v", fs.Args())
} }
settings, err := resolveAppSettings(fs, cfg.configPath, appconfig.CLIOverrides{ settings, err := resolveCommonSettings(fs, cfg.configPath, appconfig.CLIOverrides{
PromptDir: cfg.promptDirIfSet(fs), PromptDir: cfg.promptDirIfSet(fs),
ProfileDir: cfg.profileDirIfSet(fs), ProfileDir: cfg.profileDirIfSet(fs),
SchemaDir: cfg.schemaDirIfSet(fs), SchemaDir: cfg.schemaDirIfSet(fs),
@@ -368,16 +378,13 @@ func finalizeExecutionRequestConfig(fs *flag.FlagSet, cfg *runConfig) error {
return err return err
} }
cfg.promptDir = settings.PromptDir cfg.promptDir = settings.promptDir
cfg.profileDir = settings.ProfileDir cfg.profileDir = settings.profileDir
cfg.schemaDir = settings.SchemaDir cfg.schemaDir = settings.schemaDir
cfg.defaultRenderFormat = settings.DefaultRenderFormat cfg.defaultRenderFormat = settings.defaultRenderFormat
if strings.TrimSpace(cfg.promptDir) == "" { if err := validateRequiredLibraryDirs(cfg.promptDir); err != nil {
return errors.New(errPromptDirRequired) return err
}
if strings.TrimSpace(cfg.profileDir) == "" {
return errors.New(errProfileDirRequired)
} }
if strings.TrimSpace(cfg.promptID) == "" { if strings.TrimSpace(cfg.promptID) == "" {
return errors.New("--prompt is required") return errors.New("--prompt is required")
@@ -386,7 +393,9 @@ func finalizeExecutionRequestConfig(fs *flag.FlagSet, cfg *runConfig) error {
return errors.New("at least one --input is required") return errors.New("at least one --input is required")
} }
cfg.promptDir = filepath.Clean(cfg.promptDir) cfg.promptDir = filepath.Clean(cfg.promptDir)
if strings.TrimSpace(cfg.profileDir) != "" {
cfg.profileDir = filepath.Clean(cfg.profileDir) cfg.profileDir = filepath.Clean(cfg.profileDir)
}
if cfg.outputPath != "" { if cfg.outputPath != "" {
cfg.outputPath = filepath.Clean(cfg.outputPath) cfg.outputPath = filepath.Clean(cfg.outputPath)
} }
@@ -449,6 +458,34 @@ func (c *serveConfig) addrIfSet(fs *flag.FlagSet) string {
return "" return ""
} }
func (c *serveConfig) artifactRootIfSet(fs *flag.FlagSet) string {
if flagWasSet(fs, "artifact-root") {
return c.artifactRoot
}
return ""
}
func (c *serveConfig) maxRequestBytesIfSet(fs *flag.FlagSet) *int64 {
if flagWasSet(fs, "max-request-bytes") {
return &c.maxRequestBytes
}
return nil
}
func (c *serveConfig) maxArtifactBytesIfSet(fs *flag.FlagSet) *int64 {
if flagWasSet(fs, "max-artifact-bytes") {
return &c.maxArtifactBytes
}
return nil
}
func (c *serveConfig) maxResponseBytesIfSet(fs *flag.FlagSet) *int64 {
if flagWasSet(fs, "max-response-bytes") {
return &c.maxResponseBytes
}
return nil
}
func registerConfigPathFlag(fs *flag.FlagSet, target *string) { func registerConfigPathFlag(fs *flag.FlagSet, target *string) {
fs.StringVar( fs.StringVar(
target, target,
@@ -476,41 +513,81 @@ func resolveAppSettings(fs *flag.FlagSet, configPath string, overrides appconfig
return merged, nil return merged, nil
} }
func buildRunRequestFromConfig(cfg *runConfig) (domain.RunRequest, error) { func resolveCommonSettings(fs *flag.FlagSet, configPath string, overrides appconfig.CLIOverrides) (commonCommandSettings, error) {
settings, err := resolveAppSettings(fs, configPath, overrides)
if err != nil {
return commonCommandSettings{}, err
}
return commonCommandSettings{
promptDir: settings.PromptDir,
profileDir: settings.ProfileDir,
schemaDir: settings.SchemaDir,
serverAddr: settings.ServerAddr,
artifactRoot: settings.ArtifactRoot,
maxRequestBytes: settings.MaxRequestBytes,
maxArtifactBytes: settings.MaxArtifactBytes,
maxResponseBytes: settings.MaxResponseBytes,
defaultRenderFormat: settings.DefaultRenderFormat,
}, nil
}
func validateRequiredLibraryDirs(promptDir string) error {
if strings.TrimSpace(promptDir) == "" {
return errors.New(errPromptDirRequired)
}
return nil
}
func newEngine(cfg *runConfig, options ...promptkit.Option) (*promptkit.Engine, error) {
return promptkit.NewEngine(promptkit.Config{
PromptDir: cfg.promptDir,
ProfileDir: cfg.profileDir,
SchemaDir: cfg.schemaDir,
}, options...)
}
func buildRunRequestFromConfig(cfg *runConfig) (promptkit.RunRequest, error) {
inputMappings, err := parseMappings(cfg.inputRaw, false) inputMappings, err := parseMappings(cfg.inputRaw, false)
if err != nil { if err != nil {
return domain.RunRequest{}, fmt.Errorf("input parse error: %w", err) return promptkit.RunRequest{}, fmt.Errorf("input parse error: %w", err)
} }
varMappings := map[string]string{} varMappings := map[string]string{}
if len(cfg.varRaw) > 0 { if len(cfg.varRaw) > 0 {
varMappings, err = parseMappings(cfg.varRaw, false) varMappings, err = parseMappings(cfg.varRaw, false)
if err != nil { if err != nil {
return domain.RunRequest{}, fmt.Errorf("var parse error: %w", err) return promptkit.RunRequest{}, fmt.Errorf("var parse error: %w", err)
} }
} }
inputs := make(map[string]domain.ArtifactRef, len(inputMappings)) inputs := make(map[string]promptkit.ArtifactRef, len(inputMappings))
for name, path := range inputMappings { for name, path := range inputMappings {
inputs[name] = domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: path} inputs[name] = promptkit.File(path)
} }
var modelOverride *domain.ExecutionTarget var modelOverride *promptkit.ExecutionTargetOverride
if cfg.llmBaseURLSet || cfg.modelSet || cfg.temperatureSet || cfg.maxTokensSet || cfg.topPSet || cfg.apiKeyEnvSet || cfg.timeoutSet { if cfg.llmBaseURLSet || cfg.modelSet || cfg.temperatureSet || cfg.maxTokensSet || cfg.topPSet || cfg.apiKeyEnvSet || cfg.timeoutSet {
modelOverride = &domain.ExecutionTarget{ modelOverride = &promptkit.ExecutionTargetOverride{
Endpoint: cfg.llmBaseURL, Endpoint: cfg.llmBaseURL,
Model: cfg.model, Model: cfg.model,
Temperature: cfg.temperature,
MaxTokens: cfg.maxTokens,
TopP: cfg.topP,
APIKeyEnv: cfg.apiKeyEnv, APIKeyEnv: cfg.apiKeyEnv,
} }
if cfg.temperatureSet {
modelOverride.Temperature = &cfg.temperature
}
if cfg.maxTokensSet {
modelOverride.MaxTokens = &cfg.maxTokens
}
if cfg.topPSet {
modelOverride.TopP = &cfg.topP
}
if cfg.timeoutSet { if cfg.timeoutSet {
modelOverride.TimeoutSeconds = int(cfg.timeout.Seconds()) timeoutSeconds := int(cfg.timeout.Seconds())
modelOverride.TimeoutSeconds = &timeoutSeconds
} }
} }
return domain.RunRequest{ return promptkit.RunRequest{
PromptID: cfg.promptID, PromptID: cfg.promptID,
ProfileID: cfg.profileID, ProfileID: cfg.profileID,
Inputs: inputs, Inputs: inputs,
@@ -574,21 +651,21 @@ func writeOutput(stdout io.Writer, outputPath string, body []byte) error {
return os.WriteFile(outputPath, body, 0644) return os.WriteFile(outputPath, body, 0644)
} }
func determineExitCode(runErr error, result *domain.RunResult) int { func determineExitCode(runErr error, result *promptkit.RunResult) int {
if runErr != nil { if runErr != nil {
return ExitRuntimeError return ExitRuntimeError
} }
if result != nil && result.Validation.Status == domain.ValidationFailed { if result != nil && result.Validation.Status == promptkit.ValidationFailed {
return ExitValidationFailed return ExitValidationFailed
} }
return ExitOK return ExitOK
} }
func printSummary(stderr io.Writer, res *domain.RunResult) { func printSummary(stderr io.Writer, res *promptkit.RunResult) {
if res == nil { if res == nil {
return return
} }
fmt.Fprintf(stderr, "prompt=%s@%s selected_profile=%s model=%s validation=%s mode=%s validation_errors=%d prompt_hash=%s inputs=%d usage=%d/%d/%d\n", fmt.Fprintf(stderr, "prompt=%s@%s selected_profile=%s model=%s validation=%s mode=%s validation_errors=%d prompt_hash=%s inputs=%d usage=%d/%d/%d",
res.PromptID, res.PromptID,
res.PromptVersion, res.PromptVersion,
res.SelectedProfileID, res.SelectedProfileID,
@@ -602,11 +679,15 @@ func printSummary(stderr io.Writer, res *domain.RunResult) {
res.Usage.CompletionTokens, res.Usage.CompletionTokens,
res.Usage.TotalTokens, res.Usage.TotalTokens,
) )
if res.Usage.CachedTokens != 0 || res.Usage.CacheWriteTokens != 0 {
fmt.Fprintf(stderr, " cached_tokens=%d cache_write_tokens=%d", res.Usage.CachedTokens, res.Usage.CacheWriteTokens)
}
fmt.Fprintln(stderr)
} }
func printUsage(w io.Writer) { func printUsage(w io.Writer) {
fmt.Fprintln(w, "usage: scriptorium <run|render|serve> ...") fmt.Fprintln(w, "usage: scriptorium <run|render|serve> ...")
fmt.Fprintln(w, " run: scriptorium run [--config PATH] [--prompt-dir DIR] [--profile-dir DIR] --prompt ID --input name=path [--input ...] [--profile ID] [--llm-base-url URL] [--model NAME] [--api-key-env ENV] [--temperature N] [--max-tokens N] [--top-p N] [--var k=v] [--out path] [--timeout 10m]") fmt.Fprintln(w, " run: scriptorium run [--config PATH] [--prompt-dir DIR] [--profile-dir DIR] --prompt ID --input name=path [--input ...] [--profile ID] [--llm-base-url URL] [--model NAME] [--api-key-env ENV] [--temperature N] [--max-tokens N] [--top-p N] [--var k=v] [--out path] [--timeout 10m]")
fmt.Fprintln(w, " render: scriptorium render [--config PATH] [--prompt-dir DIR] [--profile-dir DIR] --prompt ID --input name=path [--input ...] [--profile ID] [--llm-base-url URL] [--model NAME] [--api-key-env ENV] [--temperature N] [--max-tokens N] [--top-p N] [--var k=v] [--format text|json] [--out path] [--timeout 10m]") fmt.Fprintln(w, " render: scriptorium render [--config PATH] [--prompt-dir DIR] [--profile-dir DIR] --prompt ID --input name=path [--input ...] [--profile ID] [--llm-base-url URL] [--model NAME] [--api-key-env ENV] [--temperature N] [--max-tokens N] [--top-p N] [--var k=v] [--format text|json] [--out path] [--timeout 10m]")
fmt.Fprintf(w, " serve: scriptorium serve [--config PATH] [--addr %s] [--prompt-dir DIR] [--profile-dir DIR] [--schema-dir DIR]\n", defaults.HTTPAddrDefault) fmt.Fprintf(w, " serve: scriptorium serve [--config PATH] [--addr %s] [--prompt-dir DIR] [--profile-dir DIR] [--schema-dir DIR] [--artifact-root DIR] [--max-request-bytes N] [--max-artifact-bytes N] [--max-response-bytes N]\n", defaults.HTTPAddrDefault)
} }

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,296 @@
package adapter_test
import (
"fmt"
"go/parser"
"go/token"
"io/fs"
"os"
"path/filepath"
"runtime"
"strconv"
"strings"
"testing"
)
const (
scriptoriumModulePath = "gitea.maximumdirect.net/eric/scriptorium"
promptkitInternalPath = "gitea.maximumdirect.net/eric/promptkit/internal"
)
var (
removedFrameworkPackageRoots = []string{
scriptoriumModulePath + "/internal/artifact",
scriptoriumModulePath + "/internal/domain",
scriptoriumModulePath + "/internal/filecatalog",
scriptoriumModulePath + "/internal/llm",
scriptoriumModulePath + "/internal/profile",
scriptoriumModulePath + "/internal/prompt",
scriptoriumModulePath + "/internal/promptdef",
scriptoriumModulePath + "/internal/usecase",
scriptoriumModulePath + "/internal/validate",
}
removedFrameworkDirectories = []string{
"internal/artifact",
"internal/domain",
"internal/filecatalog",
"internal/llm",
"internal/profile",
"internal/prompt",
"internal/promptdef",
"internal/usecase",
"internal/validate",
}
nonSourceDirectories = map[string]struct{}{
".cache": {},
".codebase-memory": {},
".git": {},
"build": {},
"coverage": {},
"dist": {},
"node_modules": {},
"out": {},
"testdata": {},
"vendor": {},
}
)
type forbiddenImport struct {
filePath string
importPath string
}
func TestApplicationBoundary(t *testing.T) {
moduleRoot := moduleRootFromTestFile(t)
violations, err := findForbiddenProductionImports(moduleRoot)
if err != nil {
t.Fatalf("scan production imports: %v", err)
}
for _, violation := range violations {
t.Errorf("%s imports forbidden package %s", violation.filePath, violation.importPath)
}
assertFrameworkImplementationAbsent(t, moduleRoot)
}
func TestForbiddenImportScannerDetectsFormerRoot(t *testing.T) {
root := t.TempDir()
sourcePath := writeGoSource(t, root, "nested/consumer/root.go", scriptoriumModulePath)
violations, err := findForbiddenProductionImports(root)
if err != nil {
t.Fatalf("scan source fixture: %v", err)
}
assertSingleViolation(t, violations, sourcePath, scriptoriumModulePath)
}
func TestForbiddenImportScannerDetectsFormerFrameworkFamily(t *testing.T) {
root := t.TempDir()
importPath := scriptoriumModulePath + "/internal/profile/builtin"
sourcePath := writeGoSource(t, root, "nested/consumer/profile.go", importPath)
violations, err := findForbiddenProductionImports(root)
if err != nil {
t.Fatalf("scan source fixture: %v", err)
}
assertSingleViolation(t, violations, sourcePath, importPath)
}
func TestForbiddenImportScannerDetectsPromptkitInternalPackages(t *testing.T) {
tests := []struct {
name string
importPath string
}{
{name: "exact internal root", importPath: promptkitInternalPath},
{name: "internal descendant", importPath: promptkitInternalPath + "/domain"},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
root := t.TempDir()
sourcePath := writeGoSource(t, root, "nested/consumer/promptkit.go", tc.importPath)
violations, err := findForbiddenProductionImports(root)
if err != nil {
t.Fatalf("scan source fixture: %v", err)
}
assertSingleViolation(t, violations, sourcePath, tc.importPath)
})
}
}
func TestForbiddenImportScannerAllowsRetainedApplicationPackages(t *testing.T) {
root := t.TempDir()
sourcePath := filepath.Join(root, "nested/consumer/application.go")
if err := os.MkdirAll(filepath.Dir(sourcePath), 0o755); err != nil {
t.Fatalf("create source fixture directory: %v", err)
}
source := `package consumer
import (
_ "gitea.maximumdirect.net/eric/promptkit"
_ "gitea.maximumdirect.net/eric/scriptorium/internal/adapter/http"
_ "gitea.maximumdirect.net/eric/scriptorium/internal/config"
_ "gitea.maximumdirect.net/eric/scriptorium/internal/defaults"
_ "gitea.maximumdirect.net/eric/scriptorium/internal/format"
)
`
if err := os.WriteFile(sourcePath, []byte(source), 0o644); err != nil {
t.Fatalf("write source fixture: %v", err)
}
violations, err := findForbiddenProductionImports(root)
if err != nil {
t.Fatalf("scan source fixture: %v", err)
}
if len(violations) != 0 {
t.Fatalf("expected retained application imports to be allowed, got %#v", violations)
}
}
func findForbiddenProductionImports(root string) ([]forbiddenImport, error) {
var violations []forbiddenImport
err := filepath.WalkDir(root, func(path string, entry fs.DirEntry, err error) error {
if err != nil {
return err
}
if entry.IsDir() {
if path != root && shouldSkipSourceDirectory(entry.Name()) {
return filepath.SkipDir
}
return nil
}
if !strings.HasSuffix(entry.Name(), ".go") || strings.HasSuffix(entry.Name(), "_test.go") {
return nil
}
file, err := parser.ParseFile(token.NewFileSet(), path, nil, parser.ImportsOnly)
if err != nil {
return fmt.Errorf("parse imports in %s: %w", path, err)
}
for _, imported := range file.Imports {
importPath, err := strconv.Unquote(imported.Path.Value)
if err != nil {
return fmt.Errorf("parse import path in %s: %w", path, err)
}
if isForbiddenProductionImport(importPath) {
violations = append(violations, forbiddenImport{
filePath: path,
importPath: importPath,
})
}
}
return nil
})
if err != nil {
return nil, fmt.Errorf("walk repository root %s: %w", root, err)
}
return violations, nil
}
func shouldSkipSourceDirectory(name string) bool {
_, skip := nonSourceDirectories[name]
return skip
}
func isForbiddenProductionImport(importPath string) bool {
if importPath == scriptoriumModulePath {
return true
}
if importPath == promptkitInternalPath || strings.HasPrefix(importPath, promptkitInternalPath+"/") {
return true
}
for _, root := range removedFrameworkPackageRoots {
if importPath == root || strings.HasPrefix(importPath, root+"/") {
return true
}
}
return false
}
func moduleRootFromTestFile(t *testing.T) string {
t.Helper()
_, testFile, _, ok := runtime.Caller(0)
if !ok {
t.Fatal("locate dependency guard source")
}
root, err := findModuleRoot(filepath.Dir(testFile))
if err != nil {
t.Fatal(err)
}
return root
}
func findModuleRoot(start string) (string, error) {
dir, err := filepath.Abs(start)
if err != nil {
return "", fmt.Errorf("resolve module search path: %w", err)
}
for {
goMod := filepath.Join(dir, "go.mod")
if info, err := os.Stat(goMod); err == nil && !info.IsDir() {
return dir, nil
} else if err != nil && !os.IsNotExist(err) {
return "", fmt.Errorf("inspect %s: %w", goMod, err)
}
parent := filepath.Dir(dir)
if parent == dir {
return "", fmt.Errorf("locate go.mod from %s", start)
}
dir = parent
}
}
func assertFrameworkImplementationAbsent(t *testing.T, moduleRoot string) {
t.Helper()
entries, err := os.ReadDir(moduleRoot)
if err != nil {
t.Fatalf("read module root: %v", err)
}
for _, entry := range entries {
if !entry.IsDir() && strings.HasSuffix(entry.Name(), ".go") && !strings.HasSuffix(entry.Name(), "_test.go") {
t.Errorf("module root contains production Go file %s", entry.Name())
}
}
for _, relativePath := range removedFrameworkDirectories {
path := filepath.Join(moduleRoot, filepath.FromSlash(relativePath))
if _, err := os.Stat(path); err == nil {
t.Errorf("removed framework directory still exists: %s", relativePath)
} else if !os.IsNotExist(err) {
t.Errorf("inspect removed framework directory %s: %v", relativePath, err)
}
}
}
func writeGoSource(t *testing.T, root, relativePath, importPath string) string {
t.Helper()
sourcePath := filepath.Join(root, filepath.FromSlash(relativePath))
if err := os.MkdirAll(filepath.Dir(sourcePath), 0o755); err != nil {
t.Fatalf("create source fixture directory: %v", err)
}
source := fmt.Sprintf("package consumer\n\nimport _ %q\n", importPath)
if err := os.WriteFile(sourcePath, []byte(source), 0o644); err != nil {
t.Fatalf("write source fixture: %v", err)
}
return sourcePath
}
func assertSingleViolation(t *testing.T, violations []forbiddenImport, sourcePath, importPath string) {
t.Helper()
if len(violations) != 1 {
t.Fatalf("expected one forbidden import, got %#v", violations)
}
if violations[0].filePath != sourcePath {
t.Fatalf("unexpected importing file: %q", violations[0].filePath)
}
if violations[0].importPath != importPath {
t.Fatalf("unexpected forbidden import: %q", violations[0].importPath)
}
}

View File

@@ -0,0 +1,165 @@
package httpadapter
import (
"context"
"crypto/sha256"
"errors"
"fmt"
"io"
"mime"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/promptkit"
)
var (
ErrFileNotAllowed = errors.New("file artifact references are not allowed")
ErrFileOutsideRoot = errors.New("file artifact path is outside artifact root")
ErrFileTooLarge = errors.New("file artifact exceeds size limit")
)
const fallbackArtifactContentType = "text/plain"
// NewRestrictedArtifactReader creates the HTTP artifact reader for a rooted
// filesystem and optional byte limit. An empty root permits inline artifacts
// but denies file references; a zero limit permits artifacts of any size.
func NewRestrictedArtifactReader(root string, maxBytes int64) (promptkit.ArtifactReader, error) {
if maxBytes < 0 {
return nil, fmt.Errorf("artifact size limit must be greater than or equal to 0")
}
cleanRoot := strings.TrimSpace(root)
if cleanRoot == "" {
return &restrictedArtifactReader{maxBytes: maxBytes}, nil
}
absRoot, err := filepath.Abs(filepath.Clean(cleanRoot))
if err != nil {
return nil, fmt.Errorf("resolve artifact root: %w", err)
}
return &restrictedArtifactReader{root: absRoot, maxBytes: maxBytes}, nil
}
type restrictedArtifactReader struct {
root string
maxBytes int64
}
var _ promptkit.ArtifactReader = (*restrictedArtifactReader)(nil)
func (r *restrictedArtifactReader) Read(ctx context.Context, ref promptkit.ArtifactRef) (*promptkit.Artifact, error) {
select {
case <-ctx.Done():
return nil, ctx.Err()
default:
}
switch ref.Type {
case promptkit.ArtifactRefInline:
return readInlineArtifact(ref)
case promptkit.ArtifactRefFile:
return r.readFileArtifact(ref)
default:
return nil, fmt.Errorf("unsupported artifact reference type %q", ref.Type)
}
}
func readInlineArtifact(ref promptkit.ArtifactRef) (*promptkit.Artifact, error) {
if ref.Body == "" {
return nil, errors.New("inline artifact body is required")
}
body := []byte(ref.Body)
return &promptkit.Artifact{
ContentType: fallbackArtifactContentType,
Body: body,
Size: int64(len(body)),
Hash: artifactHash(body),
URI: ref.URI,
}, nil
}
func (r *restrictedArtifactReader) readFileArtifact(ref promptkit.ArtifactRef) (*promptkit.Artifact, error) {
if ref.URI == "" {
return nil, errors.New("file artifact path is required")
}
if r.root == "" {
return nil, ErrFileNotAllowed
}
path, err := r.resolveLexicalPath(ref.URI)
if err != nil {
return nil, err
}
return readArtifactFile(path, r.maxBytes)
}
// resolveLexicalPath checks cleaned path containment without resolving symlinks.
func (r *restrictedArtifactReader) resolveLexicalPath(rawPath string) (string, error) {
cleanPath := filepath.Clean(strings.TrimSpace(rawPath))
candidate := cleanPath
if !filepath.IsAbs(cleanPath) {
candidate = filepath.Join(r.root, cleanPath)
}
absCandidate, err := filepath.Abs(candidate)
if err != nil {
return "", fmt.Errorf("resolve artifact path: %w", err)
}
absCandidate = filepath.Clean(absCandidate)
rel, err := filepath.Rel(r.root, absCandidate)
if err != nil {
return "", fmt.Errorf("compare artifact path to root: %w", err)
}
if rel == ".." || strings.HasPrefix(rel, ".."+string(filepath.Separator)) || filepath.IsAbs(rel) {
return "", ErrFileOutsideRoot
}
return absCandidate, nil
}
func readArtifactFile(path string, maxBytes int64) (*promptkit.Artifact, error) {
file, err := os.Open(path)
if err != nil {
return nil, fmt.Errorf("failed to read file %s: %w", path, err)
}
defer file.Close()
info, err := file.Stat()
if err != nil {
return nil, fmt.Errorf("failed to stat file %s: %w", path, err)
}
if maxBytes > 0 && info.Size() > maxBytes {
return nil, ErrFileTooLarge
}
var reader io.Reader = file
if maxBytes > 0 {
reader = io.LimitReader(file, maxBytes+1)
}
body, err := io.ReadAll(reader)
if err != nil {
return nil, fmt.Errorf("failed to read file %s: %w", path, err)
}
if maxBytes > 0 && int64(len(body)) > maxBytes {
return nil, ErrFileTooLarge
}
contentType := mime.TypeByExtension(filepath.Ext(path))
if contentType == "" {
contentType = fallbackArtifactContentType
}
return &promptkit.Artifact{
Name: filepath.Base(path),
ContentType: contentType,
Body: body,
URI: path,
Size: int64(len(body)),
Hash: artifactHash(body),
}, nil
}
func artifactHash(body []byte) string {
return fmt.Sprintf("%x", sha256.Sum256(body))
}

View File

@@ -0,0 +1,185 @@
package httpadapter
import (
"context"
"errors"
"mime"
"os"
"path/filepath"
"testing"
"gitea.maximumdirect.net/eric/promptkit"
)
func TestRestrictedArtifactReaderReadsContainedFiles(t *testing.T) {
root := t.TempDir()
outside := t.TempDir()
inputPath := filepath.Join(root, "input.html")
if err := os.WriteFile(inputPath, []byte("allowed"), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(root, "input.unknown"), []byte("unknown type"), 0o644); err != nil {
t.Fatal(err)
}
if err := os.Mkdir(filepath.Join(root, "nested"), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(outside, "secret.txt"), []byte("denied"), 0o644); err != nil {
t.Fatal(err)
}
expectedContentType := mime.TypeByExtension(filepath.Ext(inputPath))
if expectedContentType == "" {
t.Fatal("expected built-in HTML content type")
}
reader, err := NewRestrictedArtifactReader(root, 0)
if err != nil {
t.Fatalf("construct restricted reader: %v", err)
}
for _, ref := range []promptkit.ArtifactRef{
{Type: promptkit.ArtifactRefFile, URI: "nested/../input.html"},
{Type: promptkit.ArtifactRefFile, URI: inputPath},
} {
artifact, err := reader.Read(context.Background(), ref)
if err != nil {
t.Fatalf("read contained path %q: %v", ref.URI, err)
}
if artifact.Name != "input.html" || artifact.URI != inputPath || artifact.Size != int64(len("allowed")) || string(artifact.Body) != "allowed" {
t.Fatalf("unexpected artifact metadata: %#v", artifact)
}
if artifact.ContentType != expectedContentType {
t.Fatalf("unexpected artifact content type: got %q, want %q", artifact.ContentType, expectedContentType)
}
if artifact.Hash != artifactHash([]byte("allowed")) {
t.Fatalf("unexpected artifact hash: %q", artifact.Hash)
}
}
artifact, err := reader.Read(context.Background(), promptkit.File("input.unknown"))
if err != nil {
t.Fatalf("read unknown-extension path: %v", err)
}
if artifact.ContentType != fallbackArtifactContentType {
t.Fatalf("unexpected fallback content type: %q", artifact.ContentType)
}
for _, ref := range []promptkit.ArtifactRef{
{Type: promptkit.ArtifactRefFile, URI: filepath.Join("..", filepath.Base(outside), "secret.txt")},
{Type: promptkit.ArtifactRefFile, URI: filepath.Join(outside, "secret.txt")},
} {
_, err := reader.Read(context.Background(), ref)
if !errors.Is(err, ErrFileOutsideRoot) {
t.Fatalf("expected ErrFileOutsideRoot for %q, got %v", ref.URI, err)
}
}
}
func TestRestrictedArtifactReaderFollowsSymlinkAfterLexicalCheck(t *testing.T) {
root := t.TempDir()
outside := t.TempDir()
target := filepath.Join(outside, "linked.txt")
if err := os.WriteFile(target, []byte("linked outside root"), 0o644); err != nil {
t.Fatal(err)
}
if err := os.Symlink(target, filepath.Join(root, "linked.txt")); err != nil {
t.Skipf("symlink creation unavailable: %v", err)
}
reader, err := NewRestrictedArtifactReader(root, 0)
if err != nil {
t.Fatalf("construct restricted reader: %v", err)
}
artifact, err := reader.Read(context.Background(), promptkit.File("linked.txt"))
if err != nil {
t.Fatalf("read symlink inside root: %v", err)
}
if string(artifact.Body) != "linked outside root" {
t.Fatalf("unexpected symlink artifact body: %q", artifact.Body)
}
}
func TestRestrictedArtifactReaderWithoutRootDeniesFiles(t *testing.T) {
reader, err := NewRestrictedArtifactReader("", 0)
if err != nil {
t.Fatalf("construct rootless reader: %v", err)
}
artifact, err := reader.Read(context.Background(), promptkit.Inline("inline"))
if err != nil {
t.Fatalf("read inline artifact: %v", err)
}
if artifact.ContentType != fallbackArtifactContentType || string(artifact.Body) != "inline" || artifact.Hash != artifactHash([]byte("inline")) {
t.Fatalf("unexpected inline artifact: %#v", artifact)
}
_, err = reader.Read(context.Background(), promptkit.File("input.txt"))
if !errors.Is(err, ErrFileNotAllowed) {
t.Fatalf("expected ErrFileNotAllowed, got %v", err)
}
}
func TestRestrictedArtifactReaderEnforcesLimits(t *testing.T) {
root := t.TempDir()
if err := os.WriteFile(filepath.Join(root, "exact.txt"), []byte("12345"), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(root, "large.txt"), []byte("123456"), 0o644); err != nil {
t.Fatal(err)
}
reader, err := NewRestrictedArtifactReader(root, 5)
if err != nil {
t.Fatalf("construct limited reader: %v", err)
}
artifact, err := reader.Read(context.Background(), promptkit.File("exact.txt"))
if err != nil || string(artifact.Body) != "12345" {
t.Fatalf("expected exact-limit artifact, got %#v and %v", artifact, err)
}
_, err = reader.Read(context.Background(), promptkit.File("large.txt"))
if !errors.Is(err, ErrFileTooLarge) {
t.Fatalf("expected ErrFileTooLarge, got %v", err)
}
unlimited, err := NewRestrictedArtifactReader(root, 0)
if err != nil {
t.Fatalf("construct unlimited reader: %v", err)
}
artifact, err = unlimited.Read(context.Background(), promptkit.File("large.txt"))
if err != nil || string(artifact.Body) != "123456" {
t.Fatalf("expected unlimited artifact, got %#v and %v", artifact, err)
}
if _, err := NewRestrictedArtifactReader(root, -1); err == nil {
t.Fatal("expected negative limit to fail")
}
}
func TestRestrictedArtifactReaderRejectsCanceledAndMalformedReferences(t *testing.T) {
reader, err := NewRestrictedArtifactReader(t.TempDir(), 0)
if err != nil {
t.Fatalf("construct reader: %v", err)
}
canceledCtx, cancel := context.WithCancel(context.Background())
cancel()
for _, ref := range []promptkit.ArtifactRef{
promptkit.Inline("input"),
promptkit.File("input.txt"),
} {
_, err := reader.Read(canceledCtx, ref)
if !errors.Is(err, context.Canceled) {
t.Fatalf("expected cancellation for %#v, got %v", ref, err)
}
}
for _, ref := range []promptkit.ArtifactRef{
{Type: promptkit.ArtifactRefType("unsupported")},
{Type: promptkit.ArtifactRefInline},
{Type: promptkit.ArtifactRefFile},
} {
if _, err := reader.Read(context.Background(), ref); err == nil {
t.Fatalf("expected malformed reference %#v to fail", ref)
}
}
}

View File

@@ -23,13 +23,14 @@ type inputRefDTO struct {
type modelOverrideRequestDTO struct { type modelOverrideRequestDTO struct {
Endpoint string `json:"endpoint,omitempty"` Endpoint string `json:"endpoint,omitempty"`
Model string `json:"model,omitempty"` Model string `json:"model,omitempty"`
Temperature float64 `json:"temperature,omitempty"` Temperature *float64 `json:"temperature,omitempty"`
MaxTokens int `json:"max_tokens,omitempty"` MaxTokens *int `json:"max_tokens,omitempty"`
TopP float64 `json:"top_p,omitempty"` TopP *float64 `json:"top_p,omitempty"`
TimeoutSeconds int `json:"timeout_seconds,omitempty"` TimeoutSeconds *int `json:"timeout_seconds,omitempty"`
ServiceTier string `json:"service_tier,omitempty"`
ReasoningEffort string `json:"reasoning_effort,omitempty"` ReasoningEffort string `json:"reasoning_effort,omitempty"`
APIKeyEnv string `json:"api_key_env,omitempty"` APIKeyEnv string `json:"api_key_env,omitempty"`
ExtraParams map[string]string `json:"extra_params,omitempty"` ExtraParams map[string]any `json:"extra_params,omitempty"`
} }
type runResponseDTO struct { type runResponseDTO struct {
@@ -75,15 +76,18 @@ type modelParamsDTO struct {
MaxTokens int `json:"max_tokens"` MaxTokens int `json:"max_tokens"`
TopP float64 `json:"top_p"` TopP float64 `json:"top_p"`
TimeoutSeconds int `json:"timeout_seconds"` TimeoutSeconds int `json:"timeout_seconds"`
ServiceTier string `json:"service_tier,omitempty"`
ReasoningEffort string `json:"reasoning_effort,omitempty"` ReasoningEffort string `json:"reasoning_effort,omitempty"`
APIKeyEnv string `json:"api_key_env,omitempty"` APIKeyEnv string `json:"api_key_env,omitempty"`
ExtraParams map[string]string `json:"extra_params,omitempty"` ExtraParams map[string]any `json:"extra_params,omitempty"`
} }
type tokenUsageDTO struct { type tokenUsageDTO struct {
PromptTokens int `json:"prompt_tokens"` PromptTokens int `json:"prompt_tokens"`
CompletionTokens int `json:"completion_tokens"` CompletionTokens int `json:"completion_tokens"`
TotalTokens int `json:"total_tokens"` TotalTokens int `json:"total_tokens"`
CachedTokens int `json:"cached_tokens"`
CacheWriteTokens int `json:"cache_write_tokens"`
} }
type validationDTO struct { type validationDTO struct {

View File

@@ -4,25 +4,37 @@ import (
"context" "context"
"encoding/json" "encoding/json"
"errors" "errors"
"io"
"net/http" "net/http"
"strings" "strings"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/promptkit"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile" "gitea.maximumdirect.net/eric/scriptorium/internal/defaults"
"gitea.maximumdirect.net/eric/scriptorium/internal/promptdef"
"gitea.maximumdirect.net/eric/scriptorium/internal/usecase"
) )
type Runner interface { type Runner interface {
Run(ctx context.Context, req domain.RunRequest) (*domain.RunResult, error) Run(ctx context.Context, req promptkit.RunRequest) (*promptkit.RunResult, error)
} }
type Handler struct { type Handler struct {
runner Runner runner Runner
options HandlerOptions
}
type HandlerOptions struct {
MaxRequestBytes int64
MaxResponseBytes int64
} }
func NewHandler(runner Runner) *Handler { func NewHandler(runner Runner) *Handler {
return &Handler{runner: runner} return NewHandlerWithOptions(runner, HandlerOptions{
MaxRequestBytes: defaults.HTTPMaxRequestBytesDefault,
MaxResponseBytes: defaults.HTTPMaxResponseBytesDefault,
})
}
func NewHandlerWithOptions(runner Runner, options HandlerOptions) *Handler {
return &Handler{runner: runner, options: options}
} }
func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) { func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
@@ -36,9 +48,26 @@ func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
} }
var req runRequestDTO var req runRequestDTO
dec := json.NewDecoder(r.Body) body := r.Body
if h.options.MaxRequestBytes > 0 {
body = http.MaxBytesReader(w, r.Body, h.options.MaxRequestBytes)
}
dec := json.NewDecoder(body)
dec.DisallowUnknownFields() dec.DisallowUnknownFields()
if err := dec.Decode(&req); err != nil { if err := dec.Decode(&req); err != nil {
if isRequestTooLarge(err) {
writeError(w, http.StatusRequestEntityTooLarge, "request_too_large", "request body is too large")
return
}
writeError(w, http.StatusBadRequest, "invalid_json", "invalid JSON request body")
return
}
var trailing any
if err := dec.Decode(&trailing); err != io.EOF {
if isRequestTooLarge(err) {
writeError(w, http.StatusRequestEntityTooLarge, "request_too_large", "request body is too large")
return
}
writeError(w, http.StatusBadRequest, "invalid_json", "invalid JSON request body") writeError(w, http.StatusBadRequest, "invalid_json", "invalid JSON request body")
return return
} }
@@ -52,31 +81,21 @@ func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
return return
} }
mappedInputs := make(map[string]domain.ArtifactRef, len(req.Inputs)) mappedInputs := make(map[string]promptkit.ArtifactRef, len(req.Inputs))
for name, in := range req.Inputs { for name, in := range req.Inputs {
mappedInputs[name] = domain.ArtifactRef{ mappedInputs[name] = promptkit.ArtifactRef{
Type: domain.ArtifactRefType(in.Type), Type: promptkit.ArtifactRefType(in.Type),
URI: in.URI, URI: in.URI,
Body: in.Body, Body: in.Body,
} }
} }
var model *domain.ExecutionTarget var model *promptkit.ExecutionTargetOverride
if req.Model != nil { if req.Model != nil {
model = &domain.ExecutionTarget{ model = executionTargetOverrideFromModelOverrideDTO(req.Model)
Endpoint: req.Model.Endpoint,
Model: req.Model.Model,
Temperature: req.Model.Temperature,
MaxTokens: req.Model.MaxTokens,
TopP: req.Model.TopP,
TimeoutSeconds: req.Model.TimeoutSeconds,
ReasoningEffort: req.Model.ReasoningEffort,
APIKeyEnv: req.Model.APIKeyEnv,
ExtraParams: req.Model.ExtraParams,
}
} }
res, err := h.runner.Run(r.Context(), domain.RunRequest{ res, err := h.runner.Run(r.Context(), promptkit.RunRequest{
PromptID: req.PromptID, PromptID: req.PromptID,
PromptVersion: req.PromptVersion, PromptVersion: req.PromptVersion,
ProfileID: req.ProfileID, ProfileID: req.ProfileID,
@@ -109,22 +128,14 @@ func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
SelectedProfileID: res.SelectedProfileID, SelectedProfileID: res.SelectedProfileID,
ModelName: res.ModelName, ModelName: res.ModelName,
Endpoint: res.Endpoint, Endpoint: res.Endpoint,
ModelParams: modelParamsDTO{ ModelParams: modelParamsDTOFromExecutionTarget(res.EffectiveModelParams),
Endpoint: res.EffectiveModelParams.Endpoint,
Model: res.EffectiveModelParams.Model,
Temperature: res.EffectiveModelParams.Temperature,
MaxTokens: res.EffectiveModelParams.MaxTokens,
TopP: res.EffectiveModelParams.TopP,
TimeoutSeconds: res.EffectiveModelParams.TimeoutSeconds,
ReasoningEffort: res.EffectiveModelParams.ReasoningEffort,
APIKeyEnv: res.EffectiveModelParams.APIKeyEnv,
ExtraParams: res.EffectiveModelParams.ExtraParams,
},
InputHashes: res.InputHashes, InputHashes: res.InputHashes,
Usage: tokenUsageDTO{ Usage: tokenUsageDTO{
PromptTokens: res.Usage.PromptTokens, PromptTokens: res.Usage.PromptTokens,
CompletionTokens: res.Usage.CompletionTokens, CompletionTokens: res.Usage.CompletionTokens,
TotalTokens: res.Usage.TotalTokens, TotalTokens: res.Usage.TotalTokens,
CachedTokens: res.Usage.CachedTokens,
CacheWriteTokens: res.Usage.CacheWriteTokens,
}, },
StartTime: res.StartTime, StartTime: res.StartTime,
EndTime: res.EndTime, EndTime: res.EndTime,
@@ -138,10 +149,43 @@ func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
raw := res.RawOutput raw := res.RawOutput
resp.RawModelOutput = &raw resp.RawModelOutput = &raw
} }
writeJSON(w, http.StatusOK, resp) writeLimitedJSON(w, http.StatusOK, resp, h.options.MaxResponseBytes)
} }
func mapValidation(v domain.ValidationResult) validationDTO { func executionTargetOverrideFromModelOverrideDTO(dto *modelOverrideRequestDTO) *promptkit.ExecutionTargetOverride {
if dto == nil {
return nil
}
return &promptkit.ExecutionTargetOverride{
Endpoint: dto.Endpoint,
Model: dto.Model,
Temperature: dto.Temperature,
MaxTokens: dto.MaxTokens,
TopP: dto.TopP,
TimeoutSeconds: dto.TimeoutSeconds,
ServiceTier: dto.ServiceTier,
ReasoningEffort: dto.ReasoningEffort,
APIKeyEnv: dto.APIKeyEnv,
ExtraParams: dto.ExtraParams,
}
}
func modelParamsDTOFromExecutionTarget(target promptkit.ExecutionTarget) modelParamsDTO {
return modelParamsDTO{
Endpoint: target.Endpoint,
Model: target.Model,
Temperature: target.Temperature,
MaxTokens: target.MaxTokens,
TopP: target.TopP,
TimeoutSeconds: target.TimeoutSeconds,
ServiceTier: target.ServiceTier,
ReasoningEffort: target.ReasoningEffort,
APIKeyEnv: target.APIKeyEnv,
ExtraParams: target.ExtraParams,
}
}
func mapValidation(v promptkit.ValidationResult) validationDTO {
return validationDTO{ return validationDTO{
Status: string(v.Status), Status: string(v.Status),
Mode: string(v.Mode), Mode: string(v.Mode),
@@ -154,29 +198,31 @@ func mapValidation(v domain.ValidationResult) validationDTO {
func mapRunError(err error) (int, string, string) { func mapRunError(err error) (int, string, string) {
switch { switch {
case errors.Is(err, promptdef.ErrPromptDefinitionNotFound): case errors.Is(err, promptkit.ErrPromptNotFound):
return http.StatusNotFound, "prompt_not_found", "prompt definition not found" return http.StatusNotFound, "prompt_not_found", "prompt definition not found"
case errors.Is(err, profile.ErrProfileNotFound): case errors.Is(err, promptkit.ErrProfileNotFound):
return http.StatusNotFound, "profile_not_found", "execution profile not found" return http.StatusNotFound, "profile_not_found", "execution profile not found"
case errors.Is(err, promptdef.ErrInvalidYAML), errors.Is(err, promptdef.ErrInvalidPromptDefinition): case errors.Is(err, promptkit.ErrProfileRequired):
return http.StatusBadRequest, "prompt_load_failed", "failed to load prompt definition"
case errors.Is(err, profile.ErrInvalidYAML), errors.Is(err, profile.ErrInvalidProfile):
return http.StatusBadRequest, "profile_load_failed", "failed to load execution profile"
case errors.Is(err, usecase.ErrInvalidRequest) && strings.Contains(err.Error(), "profile id is required either in request or prompt default_profile"):
return http.StatusBadRequest, "profile_required", "profile_id is required when prompt default_profile is not set" return http.StatusBadRequest, "profile_required", "profile_id is required when prompt default_profile is not set"
case errors.Is(err, usecase.ErrInvalidRequest) && strings.Contains(err.Error(), "api key environment variable"): case errors.Is(err, promptkit.ErrAPIKeyEnvMissing):
return http.StatusBadRequest, "api_key_env_missing", "api_key_env is set but the environment variable is missing" return http.StatusBadRequest, "api_key_env_missing", "api_key_env is set but the environment variable is missing"
case errors.Is(err, usecase.ErrInvalidRequest): case errors.Is(err, promptkit.ErrPromptLoad):
return http.StatusBadRequest, "invalid_request", "invalid run request"
case errors.Is(err, usecase.ErrProfileLoad):
return http.StatusBadRequest, "prompt_load_failed", "failed to load prompt definition" return http.StatusBadRequest, "prompt_load_failed", "failed to load prompt definition"
case errors.Is(err, usecase.ErrArtifactLoad): case errors.Is(err, promptkit.ErrProfileLoad):
return http.StatusBadRequest, "profile_load_failed", "failed to load execution profile"
case errors.Is(err, promptkit.ErrInvalidRequest):
return http.StatusBadRequest, "invalid_request", "invalid run request"
case errors.Is(err, ErrFileNotAllowed), errors.Is(err, ErrFileOutsideRoot):
return http.StatusBadRequest, "artifact_not_allowed", "file input artifact is not allowed"
case errors.Is(err, ErrFileTooLarge):
return http.StatusRequestEntityTooLarge, "artifact_too_large", "file input artifact is too large"
case errors.Is(err, promptkit.ErrArtifactLoad):
return http.StatusBadRequest, "artifact_read_failed", "failed to read input artifact" return http.StatusBadRequest, "artifact_read_failed", "failed to read input artifact"
case errors.Is(err, usecase.ErrPromptRender): case errors.Is(err, promptkit.ErrPromptRender):
return http.StatusBadRequest, "prompt_render_failed", "failed to render prompt" return http.StatusBadRequest, "prompt_render_failed", "failed to render prompt"
case errors.Is(err, usecase.ErrLLMGenerate): case errors.Is(err, promptkit.ErrLLMGenerate):
return http.StatusBadGateway, "llm_failed", "model generation request failed" return http.StatusBadGateway, "llm_failed", "model generation request failed"
case errors.Is(err, usecase.ErrValidation): case errors.Is(err, promptkit.ErrValidation):
return http.StatusInternalServerError, "validation_runtime_failed", "validation runtime failed" return http.StatusInternalServerError, "validation_runtime_failed", "validation runtime failed"
default: default:
return http.StatusInternalServerError, "internal_error", "internal server error" return http.StatusInternalServerError, "internal_error", "internal server error"
@@ -184,9 +230,23 @@ func mapRunError(err error) (int, string, string) {
} }
func writeJSON(w http.ResponseWriter, status int, v any) { func writeJSON(w http.ResponseWriter, status int, v any) {
writeLimitedJSON(w, status, v, 0)
}
func writeLimitedJSON(w http.ResponseWriter, status int, v any, maxBytes int64) {
data, err := json.Marshal(v)
if err != nil {
writeError(w, http.StatusInternalServerError, "internal_error", "internal server error")
return
}
data = append(data, '\n')
if maxBytes > 0 && int64(len(data)) > maxBytes {
writeError(w, http.StatusRequestEntityTooLarge, "response_too_large", "response body is too large")
return
}
w.Header().Set("Content-Type", "application/json") w.Header().Set("Content-Type", "application/json")
w.WriteHeader(status) w.WriteHeader(status)
_ = json.NewEncoder(w).Encode(v) _, _ = w.Write(data)
} }
func writeError(w http.ResponseWriter, status int, code, message string) { func writeError(w http.ResponseWriter, status int, code, message string) {
@@ -197,3 +257,8 @@ func writeError(w http.ResponseWriter, status int, code, message string) {
}, },
}) })
} }
func isRequestTooLarge(err error) bool {
var maxBytesErr *http.MaxBytesError
return errors.As(err, &maxBytesErr)
}

View File

@@ -4,27 +4,26 @@ import (
"bytes" "bytes"
"context" "context"
"encoding/json" "encoding/json"
"errors"
"fmt" "fmt"
"net/http" "net/http"
"net/http/httptest" "net/http/httptest"
"os"
"path/filepath"
"reflect"
"strings" "strings"
"testing" "testing"
"time" "time"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/promptkit"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile"
"gitea.maximumdirect.net/eric/scriptorium/internal/promptdef"
"gitea.maximumdirect.net/eric/scriptorium/internal/usecase"
) )
type fakeRunner struct { type fakeRunner struct {
result *domain.RunResult result *promptkit.RunResult
err error err error
last domain.RunRequest last promptkit.RunRequest
} }
func (f *fakeRunner) Run(ctx context.Context, req domain.RunRequest) (*domain.RunResult, error) { func (f *fakeRunner) Run(ctx context.Context, req promptkit.RunRequest) (*promptkit.RunResult, error) {
f.last = req f.last = req
if f.err != nil { if f.err != nil {
return nil, f.err return nil, f.err
@@ -32,22 +31,60 @@ func (f *fakeRunner) Run(ctx context.Context, req domain.RunRequest) (*domain.Ru
return f.result, nil return f.result, nil
} }
func TestMaintainedHTTPRunExampleMatchesRequestContract(t *testing.T) {
body, err := os.ReadFile(filepath.Join("..", "..", "..", "examples", "http-run.json"))
if err != nil {
t.Fatalf("read maintained HTTP request example: %v", err)
}
runner := &fakeRunner{result: &promptkit.RunResult{}}
h := NewHandler(runner)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewReader(body))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusOK {
t.Fatalf("expected maintained HTTP request example to be accepted, got %d: %s", w.Code, w.Body.String())
}
var invalidExample map[string]json.RawMessage
if err := json.Unmarshal(body, &invalidExample); err != nil {
t.Fatalf("decode maintained HTTP request example: %v", err)
}
invalidExample["unexpected"] = json.RawMessage(`true`)
invalidBody, err := json.Marshal(invalidExample)
if err != nil {
t.Fatalf("encode structurally invalid request example: %v", err)
}
invalidReq := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewReader(invalidBody))
invalidW := httptest.NewRecorder()
h.ServeHTTP(invalidW, invalidReq)
assertHTTPErrorCode(t, invalidW, http.StatusBadRequest, "invalid_json")
}
type handlerLLMClient struct{}
func (handlerLLMClient) Generate(ctx context.Context, req promptkit.GenerateRequest) (*promptkit.GenerateResponse, error) {
return &promptkit.GenerateResponse{Content: "ok"}, nil
}
func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) { func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
start := time.Now().UTC() start := time.Now().UTC()
end := start.Add(2 * time.Second) end := start.Add(2 * time.Second)
const envName = "SCRIPTORIUM_API_KEY" const envName = "SCRIPTORIUM_API_KEY"
const secret = "never-include-me" const secret = "never-include-me"
r := &fakeRunner{result: &domain.RunResult{ r := &fakeRunner{result: &promptkit.RunResult{
RunID: "11111111-1111-4111-8111-111111111111", RunID: "11111111-1111-4111-8111-111111111111",
Artifact: domain.Artifact{ Artifact: promptkit.Artifact{
Name: "output", Name: "output",
ContentType: "text/plain", ContentType: "text/plain",
Body: []byte("hello"), Body: []byte("hello"),
Size: 5, Size: 5,
Hash: "abc", Hash: "abc",
}, },
Validation: domain.ValidationResult{Status: domain.ValidationPassed, Mode: domain.ValidationBasic, IsValid: true}, Validation: promptkit.ValidationResult{Status: promptkit.ValidationPassed, Mode: promptkit.ValidationBasic, IsValid: true},
PromptID: "prompt-1", PromptID: "prompt-1",
PromptVersion: "1.0.0", PromptVersion: "1.0.0",
PromptHash: "phash", PromptHash: "phash",
@@ -55,17 +92,24 @@ func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
SelectedProfileID: "exec-default", SelectedProfileID: "exec-default",
ModelName: "m1", ModelName: "m1",
Endpoint: "http://llm/v1", Endpoint: "http://llm/v1",
EffectiveModelParams: domain.ExecutionTarget{ EffectiveModelParams: promptkit.ExecutionTarget{
Endpoint: "http://llm/v1", Endpoint: "http://llm/v1",
Model: "m1", Model: "m1",
Temperature: 0.2, Temperature: 0.2,
MaxTokens: 42, MaxTokens: 42,
TopP: 0.9, TopP: 0.9,
TimeoutSeconds: 120, TimeoutSeconds: 120,
ServiceTier: "priority",
APIKeyEnv: envName, APIKeyEnv: envName,
}, },
InputHashes: map[string]string{"transcript": "h1"}, InputHashes: map[string]string{"transcript": "h1"},
Usage: domain.TokenUsage{PromptTokens: 1, CompletionTokens: 2, TotalTokens: 3}, Usage: promptkit.TokenUsage{
PromptTokens: 1,
CompletionTokens: 2,
TotalTokens: 3,
CachedTokens: 4,
CacheWriteTokens: 5,
},
StartTime: start, StartTime: start,
EndTime: end, EndTime: end,
Duration: 2 * time.Second, Duration: 2 * time.Second,
@@ -81,7 +125,7 @@ func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
"transcript": {"type": "file", "uri": "./t.md"} "transcript": {"type": "file", "uri": "./t.md"}
}, },
"vars": {"k": "v"}, "vars": {"k": "v"},
"model": {"model": "gpt-x", "timeout_seconds": 120, "api_key_env": "SCRIPTORIUM_API_KEY"} "model": {"model": "gpt-x", "timeout_seconds": 120, "service_tier": "flex", "api_key_env": "SCRIPTORIUM_API_KEY"}
}`) }`)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewReader(body)) req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewReader(body))
w := httptest.NewRecorder() w := httptest.NewRecorder()
@@ -110,10 +154,20 @@ func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
if metadata["model_name"] != "m1" || metadata["endpoint"] != "http://llm/v1" { if metadata["model_name"] != "m1" || metadata["endpoint"] != "http://llm/v1" {
t.Fatalf("unexpected model metadata: name=%#v endpoint=%#v", metadata["model_name"], metadata["endpoint"]) t.Fatalf("unexpected model metadata: name=%#v endpoint=%#v", metadata["model_name"], metadata["endpoint"])
} }
usage := metadata["usage"].(map[string]any)
if usage["prompt_tokens"] != float64(1) || usage["completion_tokens"] != float64(2) || usage["total_tokens"] != float64(3) {
t.Fatalf("unexpected base usage metadata: %#v", usage)
}
if usage["cached_tokens"] != float64(4) || usage["cache_write_tokens"] != float64(5) {
t.Fatalf("unexpected cache usage metadata: %#v", usage)
}
modelParams := metadata["model_params"].(map[string]any) modelParams := metadata["model_params"].(map[string]any)
if modelParams["api_key_env"] != envName { if modelParams["api_key_env"] != envName {
t.Fatalf("expected model_params.api_key_env=%q, got %#v", envName, modelParams["api_key_env"]) t.Fatalf("expected model_params.api_key_env=%q, got %#v", envName, modelParams["api_key_env"])
} }
if modelParams["service_tier"] != "priority" {
t.Fatalf("expected model_params.service_tier=priority, got %#v", modelParams["service_tier"])
}
if strings.Contains(w.Body.String(), secret) { if strings.Contains(w.Body.String(), secret) {
t.Fatalf("response leaked raw API key value: %s", w.Body.String()) t.Fatalf("response leaked raw API key value: %s", w.Body.String())
} }
@@ -130,19 +184,121 @@ func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
if r.last.Execution == nil || r.last.Execution.Model != "gpt-x" { if r.last.Execution == nil || r.last.Execution.Model != "gpt-x" {
t.Fatalf("expected model override, got %#v", r.last.Execution) t.Fatalf("expected model override, got %#v", r.last.Execution)
} }
if r.last.Execution.TimeoutSeconds != 120 { if r.last.Execution.TimeoutSeconds == nil || *r.last.Execution.TimeoutSeconds != 120 {
t.Fatalf("expected timeout_seconds override 120, got %#v", r.last.Execution) t.Fatalf("expected timeout_seconds override 120, got %#v", r.last.Execution)
} }
if r.last.Execution.ServiceTier != "flex" {
t.Fatalf("expected service_tier override flex, got %#v", r.last.Execution)
}
}
func TestHandlerInlineRefsWorkWithoutArtifactRoot(t *testing.T) {
h := newArtifactRootHandler(t, "")
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id":"p",
"inputs":{"x":{"type":"inline","body":"inline body"}}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusOK {
t.Fatalf("expected 200, got %d body=%s", w.Code, w.Body.String())
}
}
func TestHandlerFileRefsWithoutArtifactRootAreRejected(t *testing.T) {
h := newArtifactRootHandler(t, "")
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id":"p",
"inputs":{"x":{"type":"file","uri":"input.txt"}}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusBadRequest, "artifact_not_allowed")
}
func TestHandlerFileRefsUnderArtifactRootWork(t *testing.T) {
root := t.TempDir()
if err := os.WriteFile(filepath.Join(root, "input.txt"), []byte("allowed"), 0o644); err != nil {
t.Fatal(err)
}
h := newArtifactRootHandler(t, root)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id":"p",
"inputs":{"x":{"type":"file","uri":"input.txt"}}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusOK {
t.Fatalf("expected 200, got %d body=%s", w.Code, w.Body.String())
}
}
func TestHandlerFileRefsAboveArtifactLimitAreRejected(t *testing.T) {
root := t.TempDir()
if err := os.WriteFile(filepath.Join(root, "large.txt"), []byte("123456"), 0o644); err != nil {
t.Fatal(err)
}
h := newArtifactRootHandlerWithLimit(t, root, 5)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id":"p",
"inputs":{"x":{"type":"file","uri":"large.txt"}}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusRequestEntityTooLarge, "artifact_too_large")
}
func TestHandlerFileRefsOutsideArtifactRootAreRejected(t *testing.T) {
root := t.TempDir()
outside := t.TempDir()
if err := os.WriteFile(filepath.Join(outside, "secret.txt"), []byte("denied"), 0o644); err != nil {
t.Fatal(err)
}
h := newArtifactRootHandler(t, root)
tests := []struct {
name string
uri string
}{
{name: "relative traversal", uri: filepath.Join("..", filepath.Base(outside), "secret.txt")},
{name: "absolute outside root", uri: filepath.Join(outside, "secret.txt")},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
body := fmt.Sprintf(`{
"prompt_id":"p",
"inputs":{"x":{"type":"file","uri":%q}}
}`, tc.uri)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(body))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusBadRequest, "artifact_not_allowed")
})
}
} }
func TestHandlerPostRunsSuccessUsingPromptDefaultProfile(t *testing.T) { func TestHandlerPostRunsSuccessUsingPromptDefaultProfile(t *testing.T) {
r := &fakeRunner{result: &domain.RunResult{ r := &fakeRunner{result: &promptkit.RunResult{
Artifact: domain.Artifact{Body: []byte("ok")}, Artifact: promptkit.Artifact{Body: []byte("ok")},
PromptID: "prompt-1", PromptID: "prompt-1",
PromptVersion: "1.0.0", PromptVersion: "1.0.0",
SelectedProfileID: "prompt-default", SelectedProfileID: "prompt-default",
Validation: domain.ValidationResult{Status: domain.ValidationPassed, Mode: domain.ValidationBasic, IsValid: true}, Validation: promptkit.ValidationResult{Status: promptkit.ValidationPassed, Mode: promptkit.ValidationBasic, IsValid: true},
EffectiveModelParams: domain.ExecutionTarget{Endpoint: "http://llm/v1", Model: "m1"}, EffectiveModelParams: promptkit.ExecutionTarget{Endpoint: "http://llm/v1", Model: "m1"},
}} }}
h := NewHandler(r) h := NewHandler(r)
@@ -164,6 +320,265 @@ func TestHandlerPostRunsSuccessUsingPromptDefaultProfile(t *testing.T) {
if metadata["selected_profile_id"] != "prompt-default" { if metadata["selected_profile_id"] != "prompt-default" {
t.Fatalf("expected selected_profile_id from result, got %#v", metadata["selected_profile_id"]) t.Fatalf("expected selected_profile_id from result, got %#v", metadata["selected_profile_id"])
} }
usage := metadata["usage"].(map[string]any)
if usage["cached_tokens"] != float64(0) || usage["cache_write_tokens"] != float64(0) {
t.Fatalf("expected zero cache usage fields to be included, got %#v", usage)
}
}
func TestHandlerModelOverrideMapsAllSupportedExecutionFields(t *testing.T) {
r := &fakeRunner{result: &promptkit.RunResult{
Artifact: promptkit.Artifact{Body: []byte("ok")},
Validation: promptkit.ValidationResult{Status: promptkit.ValidationPassed, Mode: promptkit.ValidationBasic, IsValid: true},
EffectiveModelParams: promptkit.ExecutionTarget{Endpoint: "http://llm/v1", Model: "m1"},
}}
h := NewHandler(r)
reqBody := `{
"prompt_id": "prompt-1",
"inputs": {"transcript": {"type": "file", "uri": "./t.md"}},
"model": {
"endpoint": "http://override/v1",
"model": "override-model",
"temperature": 0.6,
"max_tokens": 250,
"top_p": 0.85,
"timeout_seconds": 33,
"service_tier": "flex",
"reasoning_effort": "medium",
"api_key_env": "SCRIPTORIUM_API_KEY",
"extra_params": {"provider_option":"on"}
}
}`
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(reqBody))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusOK {
t.Fatalf("expected 200, got %d body=%s", w.Code, w.Body.String())
}
if r.last.Execution == nil {
t.Fatalf("expected execution override in run request")
}
got := r.last.Execution
if got.Endpoint != "http://override/v1" ||
got.Model != "override-model" ||
got.ServiceTier != "flex" ||
got.ReasoningEffort != "medium" ||
got.APIKeyEnv != "SCRIPTORIUM_API_KEY" {
t.Fatalf("unexpected mapped execution target: %+v", got)
}
if got.Temperature == nil || *got.Temperature != 0.6 {
t.Fatalf("unexpected mapped temperature: %#v", got.Temperature)
}
if got.MaxTokens == nil || *got.MaxTokens != 250 {
t.Fatalf("unexpected mapped max_tokens: %#v", got.MaxTokens)
}
if got.TopP == nil || *got.TopP != 0.85 {
t.Fatalf("unexpected mapped top_p: %#v", got.TopP)
}
if got.TimeoutSeconds == nil || *got.TimeoutSeconds != 33 {
t.Fatalf("unexpected mapped timeout_seconds: %#v", got.TimeoutSeconds)
}
if !reflect.DeepEqual(got.ExtraParams, map[string]any{"provider_option": "on"}) {
t.Fatalf("unexpected mapped extra_params: %#v", got.ExtraParams)
}
}
func TestHandlerModelOverrideAcceptsJSONCompatibleExtraParams(t *testing.T) {
r := &fakeRunner{result: &promptkit.RunResult{
Artifact: promptkit.Artifact{Body: []byte("ok")},
Validation: promptkit.ValidationResult{Status: promptkit.ValidationPassed, Mode: promptkit.ValidationBasic, IsValid: true},
EffectiveModelParams: promptkit.ExecutionTarget{Endpoint: "http://llm/v1", Model: "m1"},
}}
h := NewHandler(r)
reqBody := `{
"prompt_id": "prompt-1",
"inputs": {"transcript": {"type": "file", "uri": "./t.md"}},
"model": {
"extra_params": {
"string_value": "enabled",
"number_value": 42,
"boolean_value": true,
"object_value": {"nested": "value", "count": 2},
"array_value": ["first", 3, false]
}
}
}`
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(reqBody))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusOK {
t.Fatalf("expected 200, got %d body=%s", w.Code, w.Body.String())
}
if r.last.Execution == nil {
t.Fatal("expected execution override in run request")
}
want := map[string]any{
"string_value": "enabled",
"number_value": float64(42),
"boolean_value": true,
"object_value": map[string]any{"nested": "value", "count": float64(2)},
"array_value": []any{"first", float64(3), false},
}
if !reflect.DeepEqual(r.last.Execution.ExtraParams, want) {
t.Fatalf("unexpected mapped extra_params:\ngot=%#v\nwant=%#v", r.last.Execution.ExtraParams, want)
}
}
func TestHandlerModelOverrideExplicitZeroTemperatureMapsAsPresent(t *testing.T) {
r := &fakeRunner{result: &promptkit.RunResult{
Artifact: promptkit.Artifact{Body: []byte("ok")},
Validation: promptkit.ValidationResult{Status: promptkit.ValidationPassed, Mode: promptkit.ValidationBasic, IsValid: true},
EffectiveModelParams: promptkit.ExecutionTarget{Endpoint: "http://llm/v1", Model: "m1", Temperature: 0},
}}
h := NewHandler(r)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id": "prompt-1",
"inputs": {"transcript": {"type": "file", "uri": "./t.md"}},
"model": {"temperature": 0}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusOK {
t.Fatalf("expected 200, got %d body=%s", w.Code, w.Body.String())
}
if r.last.Execution == nil || r.last.Execution.Temperature == nil {
t.Fatalf("expected temperature override to be present, got %#v", r.last.Execution)
}
if *r.last.Execution.Temperature != 0 {
t.Fatalf("expected zero temperature override, got %v", *r.last.Execution.Temperature)
}
}
func TestHandlerModelOverrideOmittedTemperatureMapsAsAbsent(t *testing.T) {
r := &fakeRunner{result: &promptkit.RunResult{
Artifact: promptkit.Artifact{Body: []byte("ok")},
Validation: promptkit.ValidationResult{Status: promptkit.ValidationPassed, Mode: promptkit.ValidationBasic, IsValid: true},
EffectiveModelParams: promptkit.ExecutionTarget{Endpoint: "http://llm/v1", Model: "m1", Temperature: 0.7},
}}
h := NewHandler(r)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id": "prompt-1",
"inputs": {"transcript": {"type": "file", "uri": "./t.md"}},
"model": {"model": "override-model"}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusOK {
t.Fatalf("expected 200, got %d body=%s", w.Code, w.Body.String())
}
if r.last.Execution == nil {
t.Fatal("expected model override")
}
if r.last.Execution.Temperature != nil {
t.Fatalf("expected omitted temperature to remain absent, got %#v", r.last.Execution.Temperature)
}
var resp map[string]any
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
t.Fatalf("invalid JSON response: %v", err)
}
metadata := resp["metadata"].(map[string]any)
params := metadata["model_params"].(map[string]any)
if params["temperature"] != 0.7 {
t.Fatalf("expected effective profile/default temperature in response, got %#v", params["temperature"])
}
}
func TestHandlerResponseMetadataModelParamsIncludesAllSupportedFields(t *testing.T) {
r := &fakeRunner{result: &promptkit.RunResult{
Artifact: promptkit.Artifact{
Name: "output",
ContentType: "text/plain",
Body: []byte("ok"),
Size: 2,
Hash: "abc",
},
Validation: promptkit.ValidationResult{Status: promptkit.ValidationPassed, Mode: promptkit.ValidationBasic, IsValid: true},
EffectiveModelParams: promptkit.ExecutionTarget{
Endpoint: "http://llm/v1",
Model: "gpt-test",
Temperature: 0.4,
MaxTokens: 321,
TopP: 0.7,
TimeoutSeconds: 45,
ServiceTier: "priority",
ReasoningEffort: "high",
APIKeyEnv: "SCRIPTORIUM_API_KEY",
ExtraParams: map[string]any{
"provider_option": "on",
"number_value": 42,
"object_value": map[string]any{"nested": "value"},
},
},
}}
h := NewHandler(r)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{"prompt_id":"p","inputs":{"x":{"type":"file","uri":"a"}}}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusOK {
t.Fatalf("expected 200, got %d body=%s", w.Code, w.Body.String())
}
var resp map[string]any
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
t.Fatalf("invalid JSON response: %v", err)
}
metadata := resp["metadata"].(map[string]any)
params := metadata["model_params"].(map[string]any)
if params["endpoint"] != "http://llm/v1" {
t.Fatalf("unexpected endpoint: %#v", params["endpoint"])
}
if params["model"] != "gpt-test" {
t.Fatalf("unexpected model: %#v", params["model"])
}
if params["temperature"] != 0.4 {
t.Fatalf("unexpected temperature: %#v", params["temperature"])
}
if params["max_tokens"] != float64(321) {
t.Fatalf("unexpected max_tokens: %#v", params["max_tokens"])
}
if params["top_p"] != 0.7 {
t.Fatalf("unexpected top_p: %#v", params["top_p"])
}
if params["timeout_seconds"] != float64(45) {
t.Fatalf("unexpected timeout_seconds: %#v", params["timeout_seconds"])
}
if params["service_tier"] != "priority" {
t.Fatalf("unexpected service_tier: %#v", params["service_tier"])
}
if params["reasoning_effort"] != "high" {
t.Fatalf("unexpected reasoning_effort: %#v", params["reasoning_effort"])
}
if params["api_key_env"] != "SCRIPTORIUM_API_KEY" {
t.Fatalf("unexpected api_key_env: %#v", params["api_key_env"])
}
extraParams, ok := params["extra_params"].(map[string]any)
if !ok {
t.Fatalf("expected extra_params object, got %#v", params["extra_params"])
}
if extraParams["provider_option"] != "on" {
t.Fatalf("unexpected extra_params.provider_option: %#v", extraParams["provider_option"])
}
if extraParams["number_value"] != float64(42) {
t.Fatalf("unexpected extra_params.number_value: %#v", extraParams["number_value"])
}
objectValue, ok := extraParams["object_value"].(map[string]any)
if !ok || objectValue["nested"] != "value" {
t.Fatalf("unexpected extra_params.object_value: %#v", extraParams["object_value"])
}
} }
func TestHandlerInvalidJSON(t *testing.T) { func TestHandlerInvalidJSON(t *testing.T) {
@@ -178,6 +593,69 @@ func TestHandlerInvalidJSON(t *testing.T) {
} }
} }
func TestHandlerRejectsTrailingJSON(t *testing.T) {
h := NewHandler(&fakeRunner{})
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{"prompt_id":"p","inputs":{"x":{"type":"file","uri":"a"}}} {}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusBadRequest, "invalid_json")
}
func TestHandlerRequestTooLarge(t *testing.T) {
h := NewHandlerWithOptions(&fakeRunner{}, HandlerOptions{MaxRequestBytes: 12})
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{"prompt_id":"p","inputs":{"x":{"type":"file","uri":"a"}}}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusRequestEntityTooLarge, "request_too_large")
}
func TestHandlerMalformedJSONBelowLimitStillBadRequest(t *testing.T) {
h := NewHandlerWithOptions(&fakeRunner{}, HandlerOptions{MaxRequestBytes: 1024})
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString("{"))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusBadRequest, "invalid_json")
}
func TestHandlerResponseTooLarge(t *testing.T) {
h := NewHandlerWithOptions(&fakeRunner{result: &promptkit.RunResult{
Artifact: promptkit.Artifact{Body: []byte(strings.Repeat("x", 128))},
Validation: promptkit.ValidationResult{Status: promptkit.ValidationPassed, Mode: promptkit.ValidationBasic, IsValid: true},
EffectiveModelParams: promptkit.ExecutionTarget{Endpoint: "http://llm/v1", Model: "m1"},
}}, HandlerOptions{MaxRequestBytes: 1024, MaxResponseBytes: 64})
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{"prompt_id":"p","inputs":{"x":{"type":"file","uri":"a"}}}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusRequestEntityTooLarge, "response_too_large")
}
func TestHandlerRawOutputDoesNotBypassResponseLimit(t *testing.T) {
h := NewHandlerWithOptions(&fakeRunner{result: &promptkit.RunResult{
Artifact: promptkit.Artifact{Body: []byte("ok")},
RawOutput: strings.Repeat("raw", 80),
Validation: promptkit.ValidationResult{Status: promptkit.ValidationPassed, Mode: promptkit.ValidationBasic, IsValid: true},
EffectiveModelParams: promptkit.ExecutionTarget{Endpoint: "http://llm/v1", Model: "m1"},
}}, HandlerOptions{MaxRequestBytes: 1024, MaxResponseBytes: 128})
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id":"p",
"inputs":{"x":{"type":"file","uri":"a"}},
"include_raw_output":true
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusRequestEntityTooLarge, "response_too_large")
}
func TestHandlerMissingPromptID(t *testing.T) { func TestHandlerMissingPromptID(t *testing.T) {
h := NewHandler(&fakeRunner{}) h := NewHandler(&fakeRunner{})
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{"inputs":{"x":{"type":"file","uri":"a"}}}`)) req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{"inputs":{"x":{"type":"file","uri":"a"}}}`))
@@ -198,7 +676,32 @@ func TestHandlerMissingPromptID(t *testing.T) {
} }
} }
func TestHandlerUsecaseErrorMapping(t *testing.T) { func TestHandlerReservedExtraParamsThroughEngineMapsToInvalidRequest(t *testing.T) {
h := NewHandler(newHandlerEngineWithDefaultClient(t))
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id":"p",
"inputs":{"x":{"type":"inline","body":"input"}},
"model":{"extra_params":{"model":"collision"}}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusBadRequest {
t.Fatalf("expected 400, got %d body=%s", w.Code, w.Body.String())
}
var resp map[string]any
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
t.Fatalf("invalid JSON response: %v", err)
}
errBody := resp["error"].(map[string]any)
if errBody["code"] != "invalid_request" {
t.Fatalf("expected invalid_request code, got %#v", errBody["code"])
}
}
func TestHandlerPublicErrorMapping(t *testing.T) {
tests := []struct { tests := []struct {
name string name string
err error err error
@@ -207,16 +710,20 @@ func TestHandlerUsecaseErrorMapping(t *testing.T) {
message string message string
avoidCause string avoidCause string
}{ }{
{name: "prompt not found", err: wrap(usecase.ErrProfileLoad, promptdef.ErrPromptDefinitionNotFound), status: http.StatusNotFound, code: "prompt_not_found", message: "prompt definition not found"}, {name: "prompt not found", err: promptkit.ErrPromptNotFound, status: http.StatusNotFound, code: "prompt_not_found", message: "prompt definition not found"},
{name: "prompt load invalid", err: wrap(usecase.ErrProfileLoad, promptdef.ErrInvalidPromptDefinition), status: http.StatusBadRequest, code: "prompt_load_failed", message: "failed to load prompt definition"}, {name: "prompt load", err: wrap(promptkit.ErrPromptLoad, fmt.Errorf("read failed")), status: http.StatusBadRequest, code: "prompt_load_failed", message: "failed to load prompt definition", avoidCause: "read failed"},
{name: "missing profile/default", err: wrap(usecase.ErrInvalidRequest, errors.New("profile id is required either in request or prompt default_profile")), status: http.StatusBadRequest, code: "profile_required", message: "profile_id is required when prompt default_profile is not set"}, {name: "missing profile/default", err: wrap(promptkit.ErrProfileRequired, promptkit.ErrInvalidRequest), status: http.StatusBadRequest, code: "profile_required", message: "profile_id is required when prompt default_profile is not set"},
{name: "profile not found", err: wrap(usecase.ErrProfileLoad, profile.ErrProfileNotFound), status: http.StatusNotFound, code: "profile_not_found", message: "execution profile not found"}, {name: "profile not found", err: promptkit.ErrProfileNotFound, status: http.StatusNotFound, code: "profile_not_found", message: "execution profile not found"},
{name: "profile invalid", err: wrap(usecase.ErrProfileLoad, profile.ErrInvalidProfile), status: http.StatusBadRequest, code: "profile_load_failed", message: "failed to load execution profile"}, {name: "profile load", err: wrap(promptkit.ErrProfileLoad, fmt.Errorf("read failed")), status: http.StatusBadRequest, code: "profile_load_failed", message: "failed to load execution profile", avoidCause: "read failed"},
{name: "api key env missing", err: wrap(usecase.ErrInvalidRequest, errors.New(`api key environment variable "SCRIPTORIUM_API_KEY" is not set`)), status: http.StatusBadRequest, code: "api_key_env_missing", message: "api_key_env is set but the environment variable is missing"}, {name: "api key env missing", err: wrap(promptkit.ErrAPIKeyEnvMissing, promptkit.ErrInvalidRequest), status: http.StatusBadRequest, code: "api_key_env_missing", message: "api_key_env is set but the environment variable is missing"},
{name: "artifact", err: wrap(usecase.ErrArtifactLoad, fmt.Errorf("read failed")), status: http.StatusBadRequest, code: "artifact_read_failed", message: "failed to read input artifact", avoidCause: "read failed"}, {name: "invalid request", err: promptkit.ErrInvalidRequest, status: http.StatusBadRequest, code: "invalid_request", message: "invalid run request"},
{name: "prompt render", err: wrap(usecase.ErrPromptRender, fmt.Errorf("render failed")), status: http.StatusBadRequest, code: "prompt_render_failed", message: "failed to render prompt", avoidCause: "render failed"}, {name: "file denied", err: ErrFileNotAllowed, status: http.StatusBadRequest, code: "artifact_not_allowed", message: "file input artifact is not allowed"},
{name: "llm", err: wrap(usecase.ErrLLMGenerate, fmt.Errorf("llm failed")), status: http.StatusBadGateway, code: "llm_failed", message: "model generation request failed", avoidCause: "llm failed"}, {name: "file outside root", err: ErrFileOutsideRoot, status: http.StatusBadRequest, code: "artifact_not_allowed", message: "file input artifact is not allowed"},
{name: "validation runtime", err: wrap(usecase.ErrValidation, fmt.Errorf("validator broke")), status: http.StatusInternalServerError, code: "validation_runtime_failed", message: "validation runtime failed", avoidCause: "validator broke"}, {name: "file too large", err: ErrFileTooLarge, status: http.StatusRequestEntityTooLarge, code: "artifact_too_large", message: "file input artifact is too large"},
{name: "artifact", err: wrap(promptkit.ErrArtifactLoad, fmt.Errorf("read failed")), status: http.StatusBadRequest, code: "artifact_read_failed", message: "failed to read input artifact", avoidCause: "read failed"},
{name: "prompt render", err: wrap(promptkit.ErrPromptRender, fmt.Errorf("render failed")), status: http.StatusBadRequest, code: "prompt_render_failed", message: "failed to render prompt", avoidCause: "render failed"},
{name: "llm", err: wrap(promptkit.ErrLLMGenerate, fmt.Errorf("llm failed")), status: http.StatusBadGateway, code: "llm_failed", message: "model generation request failed", avoidCause: "llm failed"},
{name: "validation runtime", err: wrap(promptkit.ErrValidation, fmt.Errorf("validator broke")), status: http.StatusInternalServerError, code: "validation_runtime_failed", message: "validation runtime failed", avoidCause: "validator broke"},
} }
for _, tc := range tests { for _, tc := range tests {
@@ -269,12 +776,12 @@ func TestHandlerRawAPIKeyRejectedByStrictJSON(t *testing.T) {
} }
func TestHandlerValidationFailureStillSuccessAndRawOutputOptIn(t *testing.T) { func TestHandlerValidationFailureStillSuccessAndRawOutputOptIn(t *testing.T) {
h := NewHandler(&fakeRunner{result: &domain.RunResult{ h := NewHandler(&fakeRunner{result: &promptkit.RunResult{
Artifact: domain.Artifact{Body: []byte("bad json")}, Artifact: promptkit.Artifact{Body: []byte("bad json")},
RawOutput: "bad json", RawOutput: "bad json",
Validation: domain.ValidationResult{ Validation: promptkit.ValidationResult{
Status: domain.ValidationFailed, Status: promptkit.ValidationFailed,
Mode: domain.ValidationJSON, Mode: promptkit.ValidationJSON,
Errors: []string{"invalid JSON"}, Errors: []string{"invalid JSON"},
}, },
}}) }})
@@ -318,3 +825,82 @@ func TestHandlerValidationFailureStillSuccessAndRawOutputOptIn(t *testing.T) {
func wrap(stage error, cause error) error { func wrap(stage error, cause error) error {
return fmt.Errorf("%w: %w", stage, cause) return fmt.Errorf("%w: %w", stage, cause)
} }
func newArtifactRootHandler(t *testing.T, root string) *Handler {
t.Helper()
return newArtifactRootHandlerWithLimit(t, root, 0)
}
func newArtifactRootHandlerWithLimit(t *testing.T, root string, maxArtifactBytes int64) *Handler {
t.Helper()
reader, err := NewRestrictedArtifactReader(root, maxArtifactBytes)
if err != nil {
t.Fatalf("expected restricted artifact reader: %v", err)
}
return NewHandler(newHandlerEngine(t, promptkit.WithArtifactReader(reader)))
}
func newHandlerEngine(t *testing.T, options ...promptkit.Option) *promptkit.Engine {
t.Helper()
return newHandlerEngineWithOptions(t, append(options, promptkit.WithLLMClient(handlerLLMClient{}))...)
}
func newHandlerEngineWithDefaultClient(t *testing.T, options ...promptkit.Option) *promptkit.Engine {
t.Helper()
return newHandlerEngineWithOptions(t, options...)
}
func newHandlerEngineWithOptions(t *testing.T, options ...promptkit.Option) *promptkit.Engine {
t.Helper()
promptDir := t.TempDir()
profileDir := t.TempDir()
if err := os.WriteFile(filepath.Join(promptDir, "prompt.yaml"), []byte(`id: p
version: "1"
default_profile: exec
messages:
- role: user
content: "hi"
output:
format: text
validation_mode: none
repair_attempts: 0
`), 0o644); err != nil {
t.Fatalf("write prompt fixture: %v", err)
}
if err := os.WriteFile(filepath.Join(profileDir, "profile.yaml"), []byte(`id: exec
endpoint: http://example.invalid/v1
model: model
`), 0o644); err != nil {
t.Fatalf("write profile fixture: %v", err)
}
engine, err := promptkit.NewEngine(promptkit.Config{
PromptDir: promptDir,
ProfileDir: profileDir,
}, options...)
if err != nil {
t.Fatalf("construct public engine: %v", err)
}
return engine
}
func assertHTTPErrorCode(t *testing.T, w *httptest.ResponseRecorder, status int, code string) {
t.Helper()
if w.Code != status {
t.Fatalf("expected %d, got %d body=%s", status, w.Code, w.Body.String())
}
var resp map[string]any
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
t.Fatalf("invalid JSON response: %v", err)
}
errBody := resp["error"].(map[string]any)
if errBody["code"] != code {
t.Fatalf("expected code %q, got %#v", code, errBody["code"])
}
}

View File

@@ -1,110 +0,0 @@
package artifact
import (
"context"
"crypto/sha256"
"errors"
"fmt"
"gitea.maximumdirect.net/eric/scriptorium/internal/defaults"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
"mime"
"os"
"path/filepath"
)
var (
ErrUnsupportedRefType = errors.New("unsupported artifact reference type")
ErrMissingInlineBody = errors.New("missing body for inline artifact")
ErrMissingFilePath = errors.New("missing file path for file artifact")
)
// Reader resolves artifact references into actual artifacts.
type Reader interface {
Read(ctx context.Context, ref domain.ArtifactRef) (*domain.Artifact, error)
}
// CompositeReader routes artifact resolution based on the reference type.
type CompositeReader struct {
inlineReader *inlineReader
fileReader *fileReader
}
func NewCompositeReader() Reader {
return &CompositeReader{
inlineReader: &inlineReader{},
fileReader: &fileReader{},
}
}
func (c *CompositeReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domain.Artifact, error) {
select {
case <-ctx.Done():
return nil, ctx.Err()
default:
}
switch ref.Type {
case domain.ArtifactRefInline:
return c.inlineReader.Read(ctx, ref)
case domain.ArtifactRefFile:
return c.fileReader.Read(ctx, ref)
default:
return nil, fmt.Errorf("%w: %s", ErrUnsupportedRefType, ref.Type)
}
}
type inlineReader struct{}
func (r *inlineReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domain.Artifact, error) {
select {
case <-ctx.Done():
return nil, ctx.Err()
default:
}
if ref.Body == "" {
return nil, ErrMissingInlineBody
}
body := []byte(ref.Body)
return &domain.Artifact{
ContentType: defaults.ContentTypeTextPlain,
Body: body,
Size: int64(len(body)),
Hash: fmt.Sprintf("%x", sha256.Sum256(body)),
URI: ref.URI,
}, nil
}
type fileReader struct{}
func (r *fileReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domain.Artifact, error) {
select {
case <-ctx.Done():
return nil, ctx.Err()
default:
}
if ref.URI == "" {
return nil, ErrMissingFilePath
}
data, err := os.ReadFile(ref.URI)
if err != nil {
return nil, fmt.Errorf("failed to read file %s: %w", ref.URI, err)
}
contentType := mime.TypeByExtension(filepath.Ext(ref.URI))
if contentType == "" {
contentType = defaults.ContentTypeTextPlain
}
return &domain.Artifact{
Name: filepath.Base(ref.URI),
ContentType: contentType,
Body: data,
URI: ref.URI,
Size: int64(len(data)),
Hash: fmt.Sprintf("%x", sha256.Sum256(data)),
}, nil
}

View File

@@ -1,105 +0,0 @@
package artifact
import (
"context"
"errors"
"os"
"testing"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
)
func TestCompositeReader_Read(t *testing.T) {
reader := NewCompositeReader()
ctx := context.Background()
t.Run("inline artifact", func(t *testing.T) {
ref := domain.ArtifactRef{
Type: domain.ArtifactRefInline,
Body: "hello world",
}
art, err := reader.Read(ctx, ref)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if string(art.Body) != "hello world" {
t.Errorf("expected 'hello world', got %s", string(art.Body))
}
if art.ContentType != "text/plain" {
t.Errorf("expected text/plain content type, got %q", art.ContentType)
}
if art.Hash != "b94d27b9934d3e08a52e52d7da7dabfac484efe37a5380ee9088f7ace2efcde9" {
t.Errorf("unexpected hash: %s", art.Hash)
}
})
t.Run("inline artifact missing body", func(t *testing.T) {
ref := domain.ArtifactRef{
Type: domain.ArtifactRefInline,
Body: "",
}
_, err := reader.Read(ctx, ref)
if !errors.Is(err, ErrMissingInlineBody) {
t.Errorf("expected ErrMissingInlineBody, got %v", err)
}
})
t.Run("unsupported ref type", func(t *testing.T) {
ref := domain.ArtifactRef{
Type: domain.ArtifactRefS3,
URI: "s3://bucket/key",
}
_, err := reader.Read(ctx, ref)
if !errors.Is(err, ErrUnsupportedRefType) {
t.Error("expected error for unsupported type")
}
})
}
func TestFileReader_Read(t *testing.T) {
content := []byte("test file content")
tmpFile, err := os.CreateTemp("", "artifact_test_*.txt")
if err != nil {
t.Fatal(err)
}
defer os.Remove(tmpFile.Name())
if _, err := tmpFile.Write(content); err != nil {
t.Fatal(err)
}
tmpFile.Close()
reader := NewCompositeReader()
ctx := context.Background()
t.Run("file artifact loading", func(t *testing.T) {
ref := domain.ArtifactRef{
Type: domain.ArtifactRefFile,
URI: tmpFile.Name(),
}
art, err := reader.Read(ctx, ref)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if string(art.Body) != string(content) {
t.Errorf("expected %s, got %s", string(content), string(art.Body))
}
if art.Name == "" {
t.Error("expected name to be inferred from filename")
}
if art.Hash != "60f5237ed4049f0382661ef009d2bc42e48c3ceb3edb6600f7024e7ab3b838f3" {
t.Errorf("unexpected hash: %s", art.Hash)
}
})
t.Run("missing file path", func(t *testing.T) {
ref := domain.ArtifactRef{
Type: domain.ArtifactRefFile,
URI: "",
}
_, err := reader.Read(ctx, ref)
if !errors.Is(err, ErrMissingFilePath) {
t.Errorf("expected ErrMissingFilePath, got %v", err)
}
})
}

View File

@@ -41,6 +41,10 @@ type Config struct {
type ServerConfig struct { type ServerConfig struct {
Addr string `yaml:"addr"` Addr string `yaml:"addr"`
ArtifactRoot string `yaml:"artifact_root"`
MaxRequestBytes *int64 `yaml:"max_request_bytes"`
MaxArtifactBytes *int64 `yaml:"max_artifact_bytes"`
MaxResponseBytes *int64 `yaml:"max_response_bytes"`
} }
type DefaultsConfig struct { type DefaultsConfig struct {
@@ -53,6 +57,10 @@ type AppSettings struct {
ProfileDir string ProfileDir string
SchemaDir string SchemaDir string
ServerAddr string ServerAddr string
ArtifactRoot string
MaxRequestBytes int64
MaxArtifactBytes int64
MaxResponseBytes int64
DefaultRenderFormat renderformat.PreparedRunOutputFormat DefaultRenderFormat renderformat.PreparedRunOutputFormat
} }
@@ -62,6 +70,10 @@ type CLIOverrides struct {
ProfileDir string ProfileDir string
SchemaDir string SchemaDir string
ServerAddr string ServerAddr string
ArtifactRoot string
MaxRequestBytes *int64
MaxArtifactBytes *int64
MaxResponseBytes *int64
RenderFormat string RenderFormat string
} }
@@ -70,6 +82,9 @@ func BuiltInDefaults() AppSettings {
return AppSettings{ return AppSettings{
SchemaDir: defaults.SchemaDirDefault, SchemaDir: defaults.SchemaDirDefault,
ServerAddr: defaults.HTTPAddrDefault, ServerAddr: defaults.HTTPAddrDefault,
MaxRequestBytes: defaults.HTTPMaxRequestBytesDefault,
MaxArtifactBytes: defaults.HTTPMaxArtifactBytesDefault,
MaxResponseBytes: defaults.HTTPMaxResponseBytesDefault,
DefaultRenderFormat: renderformat.DefaultPreparedRunOutputFormat, DefaultRenderFormat: renderformat.DefaultPreparedRunOutputFormat,
} }
} }
@@ -142,6 +157,27 @@ func ApplyCLIOverrides(base AppSettings, overrides CLIOverrides) (AppSettings, e
if v := strings.TrimSpace(overrides.ServerAddr); v != "" { if v := strings.TrimSpace(overrides.ServerAddr); v != "" {
out.ServerAddr = v out.ServerAddr = v
} }
if v := strings.TrimSpace(overrides.ArtifactRoot); v != "" {
out.ArtifactRoot = filepath.Clean(v)
}
if overrides.MaxRequestBytes != nil {
if *overrides.MaxRequestBytes < 0 {
return AppSettings{}, fmt.Errorf("%w: server.max_request_bytes must be greater than or equal to 0", ErrInvalidConfig)
}
out.MaxRequestBytes = *overrides.MaxRequestBytes
}
if overrides.MaxArtifactBytes != nil {
if *overrides.MaxArtifactBytes < 0 {
return AppSettings{}, fmt.Errorf("%w: server.max_artifact_bytes must be greater than or equal to 0", ErrInvalidConfig)
}
out.MaxArtifactBytes = *overrides.MaxArtifactBytes
}
if overrides.MaxResponseBytes != nil {
if *overrides.MaxResponseBytes < 0 {
return AppSettings{}, fmt.Errorf("%w: server.max_response_bytes must be greater than or equal to 0", ErrInvalidConfig)
}
out.MaxResponseBytes = *overrides.MaxResponseBytes
}
if rawFormat := strings.TrimSpace(overrides.RenderFormat); rawFormat != "" { if rawFormat := strings.TrimSpace(overrides.RenderFormat); rawFormat != "" {
parsed, err := renderformat.ParsePreparedRunOutputFormat(rawFormat) parsed, err := renderformat.ParsePreparedRunOutputFormat(rawFormat)
if err != nil { if err != nil {
@@ -181,6 +217,27 @@ func applyConfig(base AppSettings, cfg Config) (AppSettings, error) {
if v := strings.TrimSpace(cfg.Server.Addr); v != "" { if v := strings.TrimSpace(cfg.Server.Addr); v != "" {
out.ServerAddr = v out.ServerAddr = v
} }
if v := strings.TrimSpace(cfg.Server.ArtifactRoot); v != "" {
out.ArtifactRoot = filepath.Clean(v)
}
if cfg.Server.MaxRequestBytes != nil {
if *cfg.Server.MaxRequestBytes < 0 {
return AppSettings{}, fmt.Errorf("%w: server.max_request_bytes must be greater than or equal to 0", ErrInvalidConfig)
}
out.MaxRequestBytes = *cfg.Server.MaxRequestBytes
}
if cfg.Server.MaxArtifactBytes != nil {
if *cfg.Server.MaxArtifactBytes < 0 {
return AppSettings{}, fmt.Errorf("%w: server.max_artifact_bytes must be greater than or equal to 0", ErrInvalidConfig)
}
out.MaxArtifactBytes = *cfg.Server.MaxArtifactBytes
}
if cfg.Server.MaxResponseBytes != nil {
if *cfg.Server.MaxResponseBytes < 0 {
return AppSettings{}, fmt.Errorf("%w: server.max_response_bytes must be greater than or equal to 0", ErrInvalidConfig)
}
out.MaxResponseBytes = *cfg.Server.MaxResponseBytes
}
if rawFormat := strings.TrimSpace(cfg.Defaults.RenderFormat); rawFormat != "" { if rawFormat := strings.TrimSpace(cfg.Defaults.RenderFormat); rawFormat != "" {
parsed, err := renderformat.ParsePreparedRunOutputFormat(rawFormat) parsed, err := renderformat.ParsePreparedRunOutputFormat(rawFormat)
if err != nil { if err != nil {

View File

@@ -6,6 +6,7 @@ import (
"path/filepath" "path/filepath"
"testing" "testing"
"gitea.maximumdirect.net/eric/scriptorium/internal/defaults"
renderformat "gitea.maximumdirect.net/eric/scriptorium/internal/format" renderformat "gitea.maximumdirect.net/eric/scriptorium/internal/format"
) )
@@ -24,6 +25,20 @@ func TestLoadConfigMissingImplicitPathUsesBuiltInDefaults(t *testing.T) {
} }
} }
func TestBuiltInDefaultsIncludeHTTPSizeLimits(t *testing.T) {
got := BuiltInDefaults()
if got.MaxRequestBytes != defaults.HTTPMaxRequestBytesDefault {
t.Fatalf("unexpected max request bytes: %d", got.MaxRequestBytes)
}
if got.MaxArtifactBytes != defaults.HTTPMaxArtifactBytesDefault {
t.Fatalf("unexpected max artifact bytes: %d", got.MaxArtifactBytes)
}
if got.MaxResponseBytes != defaults.HTTPMaxResponseBytesDefault {
t.Fatalf("unexpected max response bytes: %d", got.MaxResponseBytes)
}
}
func TestLoadConfigMissingExplicitPathReturnsError(t *testing.T) { func TestLoadConfigMissingExplicitPathReturnsError(t *testing.T) {
tmp := t.TempDir() tmp := t.TempDir()
missing := filepath.Join(tmp, "missing.yml") missing := filepath.Join(tmp, "missing.yml")
@@ -92,6 +107,10 @@ profile_dir: ./profiles
schema_dir: ./schemas schema_dir: ./schemas
server: server:
addr: 127.0.0.1:9090 addr: 127.0.0.1:9090
artifact_root: ./artifacts
max_request_bytes: 1024
max_artifact_bytes: 2048
max_response_bytes: 4096
defaults: defaults:
render_format: json render_format: json
`) `)
@@ -113,11 +132,61 @@ defaults:
if got.ServerAddr != "127.0.0.1:9090" { if got.ServerAddr != "127.0.0.1:9090" {
t.Fatalf("unexpected server.addr: %q", got.ServerAddr) t.Fatalf("unexpected server.addr: %q", got.ServerAddr)
} }
if got.ArtifactRoot != filepath.Clean("./artifacts") {
t.Fatalf("unexpected server.artifact_root: %q", got.ArtifactRoot)
}
if got.MaxRequestBytes != 1024 {
t.Fatalf("unexpected server.max_request_bytes: %d", got.MaxRequestBytes)
}
if got.MaxArtifactBytes != 2048 {
t.Fatalf("unexpected server.max_artifact_bytes: %d", got.MaxArtifactBytes)
}
if got.MaxResponseBytes != 4096 {
t.Fatalf("unexpected server.max_response_bytes: %d", got.MaxResponseBytes)
}
if got.DefaultRenderFormat != renderformat.PreparedRunFormatJSON { if got.DefaultRenderFormat != renderformat.PreparedRunFormatJSON {
t.Fatalf("unexpected defaults.render_format: %q", got.DefaultRenderFormat) t.Fatalf("unexpected defaults.render_format: %q", got.DefaultRenderFormat)
} }
} }
func TestLoadConfigAcceptsZeroHTTPSizeLimits(t *testing.T) {
path := writeConfigFile(t, "config.yml", `
server:
max_request_bytes: 0
max_artifact_bytes: 0
max_response_bytes: 0
`)
got, err := LoadConfig(path, true)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if got.MaxRequestBytes != 0 || got.MaxArtifactBytes != 0 || got.MaxResponseBytes != 0 {
t.Fatalf("expected zero limits to be preserved, got request=%d artifact=%d response=%d", got.MaxRequestBytes, got.MaxArtifactBytes, got.MaxResponseBytes)
}
}
func TestLoadConfigRejectsNegativeHTTPSizeLimits(t *testing.T) {
tests := []struct {
name string
body string
}{
{name: "request", body: "server:\n max_request_bytes: -1\n"},
{name: "artifact", body: "server:\n max_artifact_bytes: -1\n"},
{name: "response", body: "server:\n max_response_bytes: -1\n"},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
path := writeConfigFile(t, "config.yml", tc.body)
_, err := LoadConfig(path, true)
if !errors.Is(err, ErrInvalidConfig) {
t.Fatalf("expected ErrInvalidConfig, got %v", err)
}
})
}
}
func TestLoadConfigEmptyFileResolvesToBuiltInDefaults(t *testing.T) { func TestLoadConfigEmptyFileResolvesToBuiltInDefaults(t *testing.T) {
path := writeConfigFile(t, "config.yml", "") path := writeConfigFile(t, "config.yml", "")
@@ -185,14 +254,25 @@ func TestApplyCLIOverridesAppliesPrecedence(t *testing.T) {
ProfileDir: "/from/config/profiles", ProfileDir: "/from/config/profiles",
SchemaDir: "/from/config/schemas", SchemaDir: "/from/config/schemas",
ServerAddr: ":1234", ServerAddr: ":1234",
ArtifactRoot: "/from/config/artifacts",
MaxRequestBytes: 111,
MaxArtifactBytes: 222,
MaxResponseBytes: 333,
DefaultRenderFormat: renderformat.PreparedRunFormatJSON, DefaultRenderFormat: renderformat.PreparedRunFormatJSON,
} }
maxRequestBytes := int64(0)
maxArtifactBytes := int64(444)
maxResponseBytes := int64(555)
got, err := ApplyCLIOverrides(base, CLIOverrides{ got, err := ApplyCLIOverrides(base, CLIOverrides{
PromptDir: "./prompts-cli", PromptDir: "./prompts-cli",
ProfileDir: "./profiles-cli", ProfileDir: "./profiles-cli",
SchemaDir: "./schemas-cli", SchemaDir: "./schemas-cli",
ServerAddr: ":8081", ServerAddr: ":8081",
ArtifactRoot: "./artifacts-cli",
MaxRequestBytes: &maxRequestBytes,
MaxArtifactBytes: &maxArtifactBytes,
MaxResponseBytes: &maxResponseBytes,
RenderFormat: "text", RenderFormat: "text",
}) })
if err != nil { if err != nil {
@@ -211,11 +291,45 @@ func TestApplyCLIOverridesAppliesPrecedence(t *testing.T) {
if got.ServerAddr != ":8081" { if got.ServerAddr != ":8081" {
t.Fatalf("unexpected server addr: %q", got.ServerAddr) t.Fatalf("unexpected server addr: %q", got.ServerAddr)
} }
if got.ArtifactRoot != filepath.Clean("./artifacts-cli") {
t.Fatalf("unexpected artifact root: %q", got.ArtifactRoot)
}
if got.MaxRequestBytes != 0 {
t.Fatalf("unexpected max request bytes: %d", got.MaxRequestBytes)
}
if got.MaxArtifactBytes != 444 {
t.Fatalf("unexpected max artifact bytes: %d", got.MaxArtifactBytes)
}
if got.MaxResponseBytes != 555 {
t.Fatalf("unexpected max response bytes: %d", got.MaxResponseBytes)
}
if got.DefaultRenderFormat != renderformat.PreparedRunFormatText { if got.DefaultRenderFormat != renderformat.PreparedRunFormatText {
t.Fatalf("unexpected render format: %q", got.DefaultRenderFormat) t.Fatalf("unexpected render format: %q", got.DefaultRenderFormat)
} }
} }
func TestApplyCLIOverridesRejectsNegativeHTTPSizeLimits(t *testing.T) {
negative := int64(-1)
tests := []struct {
name string
overrides CLIOverrides
}{
{name: "request", overrides: CLIOverrides{MaxRequestBytes: &negative}},
{name: "artifact", overrides: CLIOverrides{MaxArtifactBytes: &negative}},
{name: "response", overrides: CLIOverrides{MaxResponseBytes: &negative}},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
_, err := ApplyCLIOverrides(BuiltInDefaults(), tc.overrides)
if !errors.Is(err, ErrInvalidConfig) {
t.Fatalf("expected ErrInvalidConfig, got %v", err)
}
})
}
}
func TestApplyCLIOverridesInvalidRenderFormatReturnsError(t *testing.T) { func TestApplyCLIOverridesInvalidRenderFormatReturnsError(t *testing.T) {
_, err := ApplyCLIOverrides(BuiltInDefaults(), CLIOverrides{RenderFormat: "yaml"}) _, err := ApplyCLIOverrides(BuiltInDefaults(), CLIOverrides{RenderFormat: "yaml"})
if err == nil { if err == nil {

View File

@@ -1,36 +1,13 @@
package defaults package defaults
import ( import "time"
"time"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
)
const ( const (
HTTPAddrDefault = ":8080" HTTPAddrDefault = ":8080"
SchemaDirDefault = "." SchemaDirDefault = "."
OutputArtifactName = "output" HTTPMaxRequestBytesDefault = 16 * 1024 * 1024
ContentTypeTextPlain = "text/plain" HTTPMaxArtifactBytesDefault = 16 * 1024 * 1024
ContentTypeTextMarkdown = "text/markdown" HTTPMaxResponseBytesDefault = 16 * 1024 * 1024
ContentTypeApplicationJSON = "application/json"
OpenAIChatCompletionsPath = "/chat/completions"
ExecutionDefaultTemperature = 0.0
ExecutionDefaultMaxTokens = 0
ExecutionDefaultTopP = 1.0
ExecutionDefaultTimeoutSeconds = 600
) )
var ( var HTTPReadHeaderTimeoutDefault = 10 * time.Second
LLMRequestTimeoutDefault = 10 * time.Minute
HTTPReadHeaderTimeoutDefault = 10 * time.Second
)
func ExecutionTargetDefault() domain.ExecutionTarget {
return domain.ExecutionTarget{
Temperature: ExecutionDefaultTemperature,
MaxTokens: ExecutionDefaultMaxTokens,
TopP: ExecutionDefaultTopP,
TimeoutSeconds: ExecutionDefaultTimeoutSeconds,
}
}

View File

@@ -1,254 +0,0 @@
package domain
import (
"time"
)
// ArtifactRefType defines how an artifact is referenced.
type ArtifactRefType string
const (
ArtifactRefInline ArtifactRefType = "inline"
ArtifactRefFile ArtifactRefType = "file"
ArtifactRefS3 ArtifactRefType = "s3"
)
// OutputFormat defines the desired format of the generated artifact.
type OutputFormat string
const (
FormatText OutputFormat = "text"
FormatMarkdown OutputFormat = "markdown"
FormatJSON OutputFormat = "json"
)
// ValidationMode defines how the output should be validated.
type ValidationMode string
const (
ValidationNone ValidationMode = "none"
ValidationBasic ValidationMode = "basic"
ValidationJSON ValidationMode = "json"
ValidationJSONSchema ValidationMode = "json_schema"
)
// ValidationStatus defines the result of a validation check.
type ValidationStatus string
const (
ValidationPassed ValidationStatus = "passed"
ValidationFailed ValidationStatus = "failed"
ValidationSkipped ValidationStatus = "skipped"
)
// RunRequest represents a request to generate a single artifact.
type RunRequest struct {
PromptID string
PromptVersion string
ProfileID string
Inputs map[string]ArtifactRef
Vars map[string]string
Execution *ExecutionTarget
Validation *OutputContract
Metadata map[string]string
}
// RunResult represents the complete result of a prompt execution run.
type RunResult struct {
RunID string
Artifact Artifact
RawOutput string
Validation ValidationResult
PromptID string
PromptVersion string
PromptHash string
RenderedPromptHash string
SelectedProfileID string
ModelName string
Endpoint string
EffectiveModelParams ExecutionTarget
InputHashes map[string]string
Usage TokenUsage
StartTime time.Time
EndTime time.Time
Duration time.Duration
Error error
}
// PreparedRun contains pre-LLM execution state from the prepare/render phase.
// It must never include resolved API key values, model output, or validation data.
type PreparedRun struct {
PromptID string `json:"prompt_id"`
PromptVersion string `json:"prompt_version,omitempty"`
PromptHash string `json:"prompt_hash,omitempty"`
SelectedProfileID string `json:"selected_profile_id"`
EffectiveModelParams ExecutionTarget `json:"effective_model_params"`
OutputContract OutputContract `json:"output_contract"`
StructuredOutput *StructuredOutputSpec `json:"structured_output,omitempty"`
InputHashes map[string]string `json:"input_hashes,omitempty"`
RenderedPromptHash string `json:"rendered_prompt_hash"`
Messages []RenderedMessage `json:"messages"`
StartTime time.Time `json:"start_time,omitempty"`
EndTime time.Time `json:"end_time,omitempty"`
DurationMS int64 `json:"duration_ms,omitempty"`
}
// ArtifactRef represents a reference to an input artifact.
type ArtifactRef struct {
Type ArtifactRefType
URI string
Body string // Used for inline
}
// Artifact represents the actual loaded content of a reference.
type Artifact struct {
Name string
ContentType string
Body []byte
URI string
Size int64
Hash string
}
// PromptDefinition represents a configured prompt execution definition.
type PromptDefinition struct {
ID string `yaml:"id"`
Version string `yaml:"version"`
DefaultProfile string `yaml:"default_profile"`
Description string `yaml:"description"`
Inputs []PromptInput `yaml:"inputs"`
Templates []PromptMessageTemplate `yaml:"templates"`
OutputFormat OutputFormat `yaml:"output_format"`
Validation OutputContract `yaml:"validation"`
}
// PromptInput describes one named input expected by a prompt definition.
type PromptInput struct {
Name string `yaml:"name"`
Required bool `yaml:"required"`
ContentType string `yaml:"content_type"`
Description string `yaml:"description"`
}
// PromptMessageTemplate defines a template for a chat message.
type PromptMessageTemplate struct {
Role string `yaml:"role"`
Content string `yaml:"content"`
ContentFile string `yaml:"content_file"`
}
// ExecutionProfile describes how and where to execute a model.
type ExecutionProfile struct {
ID string `yaml:"id"`
Endpoint string `yaml:"endpoint"`
Model string `yaml:"model"`
Temperature float64 `yaml:"temperature"`
MaxTokens int `yaml:"max_tokens"`
TopP float64 `yaml:"top_p"`
TimeoutSeconds int `yaml:"timeout_seconds"`
ReasoningEffort string `yaml:"reasoning_effort"`
APIKeyEnv string `yaml:"api_key_env"`
ExtraParams map[string]string `yaml:"extra_params"`
}
// ExecutionTarget represents effective model runtime settings for a run.
type ExecutionTarget struct {
Endpoint string `yaml:"endpoint" json:"endpoint"`
Model string `yaml:"model" json:"model"`
Temperature float64 `yaml:"temperature" json:"temperature"`
MaxTokens int `yaml:"max_tokens" json:"max_tokens"`
TopP float64 `yaml:"top_p" json:"top_p"`
TimeoutSeconds int `yaml:"timeout_seconds" json:"timeout_seconds"`
ReasoningEffort string `yaml:"reasoning_effort" json:"reasoning_effort"`
APIKeyEnv string `yaml:"api_key_env" json:"api_key_env"`
ExtraParams map[string]string `yaml:"extra_params" json:"extra_params"`
}
// OutputContract defines the requirements for the output artifact.
type OutputContract struct {
Format OutputFormat `yaml:"format"`
ValidationMode ValidationMode `yaml:"validation_mode"`
SchemaPath string `yaml:"schema_path"`
RepairAttempts int `yaml:"repair_attempts"`
}
// RenderedPrompt represents the prompt after template application.
type RenderedPrompt struct {
Messages []RenderedMessage `json:"messages"`
}
// RenderedMessage is a single message in a rendered prompt.
type RenderedMessage struct {
Role string `json:"role"`
Content string `json:"content"`
}
// GenerateRequest is the internal request passed to the LLM client.
type GenerateRequest struct {
Prompt RenderedPrompt
Target ExecutionTarget
StructuredOutput *StructuredOutputSpec
}
// StructuredOutputType indicates which provider-level output mode is requested.
type StructuredOutputType string
const (
StructuredOutputJSONSchema StructuredOutputType = "json_schema"
)
// StructuredOutputSpec describes provider-level structured output requirements.
type StructuredOutputSpec struct {
Type StructuredOutputType `json:"type"`
JSONSchema *StructuredOutputJSONSpec `json:"json_schema,omitempty"`
}
// StructuredOutputJSONSpec contains json_schema output constraints.
type StructuredOutputJSONSpec struct {
Name string `json:"name"`
Strict bool `json:"strict"`
Schema any `json:"schema"`
}
// GenerateResponse is the response received from the LLM client.
type GenerateResponse struct {
Content string
Usage TokenUsage
}
// TokenUsage tracks token consumption.
type TokenUsage struct {
PromptTokens int
CompletionTokens int
TotalTokens int
}
// ValidationResult represents the outcome of an output validation.
type ValidationResult struct {
Status ValidationStatus
Mode ValidationMode
Errors []string
SchemaPath string
RepairAttempts int
IsValid bool
}
// RunMetadata contains auditing information for a run.
type RunMetadata struct {
RunID string
PromptID string
PromptVersion string
PromptHash string
RenderedPromptHash string
SelectedProfileID string
InputHashes map[string]string
ModelEndpoint string
ModelName string
Params ExecutionTarget
Timestamp time.Time
Duration time.Duration
Usage TokenUsage
ValidationMode ValidationMode
ValidationStatus ValidationStatus
RepairAttempts int
}

View File

@@ -1,55 +0,0 @@
package domain
import (
"encoding/json"
"strings"
"testing"
)
func TestPreparedRunJSONDoesNotIncludeSecretValues(t *testing.T) {
const envName = "SCRIPTORIUM_TEST_API_KEY"
const secret = "super-secret-value"
t.Setenv(envName, secret)
prepared := PreparedRun{
PromptID: "prompt.id",
PromptVersion: "v1",
PromptHash: "prompt-hash",
SelectedProfileID: "local-fast",
EffectiveModelParams: ExecutionTarget{
Endpoint: "http://llm/v1",
Model: "gpt-test",
APIKeyEnv: envName,
},
InputHashes: map[string]string{"transcript": "hash-1"},
RenderedPromptHash: "rendered-hash",
Messages: []RenderedMessage{
{Role: "system", Content: "You are helpful."},
{Role: "user", Content: "Summarize this."},
},
}
b, err := json.Marshal(prepared)
if err != nil {
t.Fatalf("marshal failed: %v", err)
}
out := string(b)
if strings.Contains(out, secret) {
t.Fatalf("prepared run JSON unexpectedly contains secret value: %s", out)
}
if !strings.Contains(out, `"api_key_env":"`+envName+`"`) {
t.Fatalf("prepared run JSON should include api_key_env name: %s", out)
}
var top map[string]any
if err := json.Unmarshal(b, &top); err != nil {
t.Fatalf("unmarshal failed: %v", err)
}
for _, forbidden := range []string{"raw_output", "validation", "artifact"} {
if _, ok := top[forbidden]; ok {
t.Fatalf("prepared run JSON should not include %q", forbidden)
}
}
}

View File

@@ -1,4 +1,4 @@
// Package format formats already-prepared domain data for adapters. // Package format formats already-prepared public data for adapters.
package format package format
import ( import (
@@ -9,7 +9,7 @@ import (
"sort" "sort"
"strings" "strings"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/promptkit"
) )
var ErrUnknownPreparedRunFormat = errors.New("unknown prepared run format") var ErrUnknownPreparedRunFormat = errors.New("unknown prepared run format")
@@ -26,7 +26,7 @@ const (
// PreparedRunFormatter serializes a prepared run without performing use case work. // PreparedRunFormatter serializes a prepared run without performing use case work.
type PreparedRunFormatter interface { type PreparedRunFormatter interface {
Format(prepared *domain.PreparedRun) ([]byte, error) Format(prepared *promptkit.PreparedRun) ([]byte, error)
} }
// ParsePreparedRunOutputFormat parses a format name. // ParsePreparedRunOutputFormat parses a format name.
@@ -56,7 +56,7 @@ func FormatterForPreparedRun(outputFormat PreparedRunOutputFormat) (PreparedRunF
} }
// FormatPreparedRun formats a prepared run using the selected format. // FormatPreparedRun formats a prepared run using the selected format.
func FormatPreparedRun(prepared *domain.PreparedRun, outputFormat PreparedRunOutputFormat) ([]byte, error) { func FormatPreparedRun(prepared *promptkit.PreparedRun, outputFormat PreparedRunOutputFormat) ([]byte, error) {
formatter, err := FormatterForPreparedRun(outputFormat) formatter, err := FormatterForPreparedRun(outputFormat)
if err != nil { if err != nil {
return nil, err return nil, err
@@ -65,7 +65,7 @@ func FormatPreparedRun(prepared *domain.PreparedRun, outputFormat PreparedRunOut
} }
// FormatPreparedRunByName parses a format name and formats a prepared run. // FormatPreparedRunByName parses a format name and formats a prepared run.
func FormatPreparedRunByName(prepared *domain.PreparedRun, rawFormat string) ([]byte, error) { func FormatPreparedRunByName(prepared *promptkit.PreparedRun, rawFormat string) ([]byte, error) {
outputFormat, err := ParsePreparedRunOutputFormat(rawFormat) outputFormat, err := ParsePreparedRunOutputFormat(rawFormat)
if err != nil { if err != nil {
return nil, err return nil, err
@@ -75,7 +75,7 @@ func FormatPreparedRunByName(prepared *domain.PreparedRun, rawFormat string) ([]
type jsonPreparedRunFormatter struct{} type jsonPreparedRunFormatter struct{}
func (jsonPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, error) { func (jsonPreparedRunFormatter) Format(prepared *promptkit.PreparedRun) ([]byte, error) {
if prepared == nil { if prepared == nil {
return nil, errors.New("prepared run is nil") return nil, errors.New("prepared run is nil")
} }
@@ -84,7 +84,7 @@ func (jsonPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, er
type textPreparedRunFormatter struct{} type textPreparedRunFormatter struct{}
func (textPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, error) { func (textPreparedRunFormatter) Format(prepared *promptkit.PreparedRun) ([]byte, error) {
if prepared == nil { if prepared == nil {
return nil, errors.New("prepared run is nil") return nil, errors.New("prepared run is nil")
} }
@@ -96,6 +96,9 @@ func (textPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, er
if prepared.PromptHash != "" { if prepared.PromptHash != "" {
fmt.Fprintf(&b, "prompt_hash: %s\n", prepared.PromptHash) fmt.Fprintf(&b, "prompt_hash: %s\n", prepared.PromptHash)
} }
if prepared.SessionID != "" {
fmt.Fprintf(&b, "session_id: %s\n", prepared.SessionID)
}
fmt.Fprintf(&b, "rendered_prompt_hash: %s\n", prepared.RenderedPromptHash) fmt.Fprintf(&b, "rendered_prompt_hash: %s\n", prepared.RenderedPromptHash)
target := prepared.EffectiveModelParams target := prepared.EffectiveModelParams
@@ -106,6 +109,9 @@ func (textPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, er
fmt.Fprintf(&b, " max_tokens: %d\n", target.MaxTokens) fmt.Fprintf(&b, " max_tokens: %d\n", target.MaxTokens)
fmt.Fprintf(&b, " top_p: %g\n", target.TopP) fmt.Fprintf(&b, " top_p: %g\n", target.TopP)
fmt.Fprintf(&b, " timeout_seconds: %d\n", target.TimeoutSeconds) fmt.Fprintf(&b, " timeout_seconds: %d\n", target.TimeoutSeconds)
if target.ServiceTier != "" {
fmt.Fprintf(&b, " service_tier: %s\n", target.ServiceTier)
}
if target.ReasoningEffort != "" { if target.ReasoningEffort != "" {
fmt.Fprintf(&b, " reasoning_effort: %s\n", target.ReasoningEffort) fmt.Fprintf(&b, " reasoning_effort: %s\n", target.ReasoningEffort)
} }
@@ -120,7 +126,11 @@ func (textPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, er
} }
sort.Strings(keys) sort.Strings(keys)
for _, k := range keys { for _, k := range keys {
fmt.Fprintf(&b, " %s: %s\n", k, target.ExtraParams[k]) renderedValue, err := formatExtraParamTextValue(target.ExtraParams[k])
if err != nil {
return nil, fmt.Errorf("failed to format extra_params.%s: %w", k, err)
}
fmt.Fprintf(&b, " %s: %s\n", k, renderedValue)
} }
} }
@@ -136,7 +146,7 @@ func (textPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, er
fmt.Fprintln(&b, "messages:") fmt.Fprintln(&b, "messages:")
roleOrder := make([]string, 0) roleOrder := make([]string, 0)
byRole := make(map[string][]domain.RenderedMessage) byRole := make(map[string][]promptkit.RenderedMessage)
for _, msg := range prepared.Messages { for _, msg := range prepared.Messages {
if _, exists := byRole[msg.Role]; !exists { if _, exists := byRole[msg.Role]; !exists {
roleOrder = append(roleOrder, msg.Role) roleOrder = append(roleOrder, msg.Role)
@@ -148,6 +158,13 @@ func (textPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, er
messages := byRole[role] messages := byRole[role]
for i, msg := range messages { for i, msg := range messages {
fmt.Fprintf(&b, " - message: %d\n", i+1) fmt.Fprintf(&b, " - message: %d\n", i+1)
if msg.CacheControl != nil {
fmt.Fprintf(&b, " cache_control: %s", msg.CacheControl.Type)
if msg.CacheControl.TTL != "" {
fmt.Fprintf(&b, " ttl=%s", msg.CacheControl.TTL)
}
fmt.Fprintln(&b)
}
fmt.Fprintln(&b, " content: |") fmt.Fprintln(&b, " content: |")
content := msg.Content content := msg.Content
if content == "" { if content == "" {
@@ -162,3 +179,15 @@ func (textPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, er
return b.Bytes(), nil return b.Bytes(), nil
} }
func formatExtraParamTextValue(value any) (string, error) {
if s, ok := value.(string); ok {
return s, nil
}
b, err := json.Marshal(value)
if err != nil {
return "", err
}
return string(b), nil
}

View File

@@ -6,7 +6,7 @@ import (
"strings" "strings"
"testing" "testing"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/promptkit"
) )
func TestTextFormatterIncludesPreparedRunDetails(t *testing.T) { func TestTextFormatterIncludesPreparedRunDetails(t *testing.T) {
@@ -28,6 +28,7 @@ func TestTextFormatterIncludesPreparedRunDetails(t *testing.T) {
"max_tokens: 256", "max_tokens: 256",
"top_p: 0.8", "top_p: 0.8",
"timeout_seconds: 45", "timeout_seconds: 45",
"service_tier: priority",
"reasoning_effort: medium", "reasoning_effort: medium",
"api_key_env: SCRIPTORIUM_API_KEY", "api_key_env: SCRIPTORIUM_API_KEY",
"prompt_hash: prompt-hash", "prompt_hash: prompt-hash",
@@ -48,6 +49,36 @@ func TestTextFormatterIncludesPreparedRunDetails(t *testing.T) {
} }
} }
func TestTextFormatterRendersExtraParamsDeterministically(t *testing.T) {
prepared := samplePreparedRun()
prepared.EffectiveModelParams.ExtraParams = map[string]any{
"z_string": "enabled",
"b_number": 42,
"a_object": map[string]any{
"nested": "value",
"count": 2,
},
"c_array": []any{"first", 3, false},
}
out, err := FormatPreparedRun(prepared, PreparedRunFormatText)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
s := string(out)
want := strings.Join([]string{
" extra_params:",
" a_object: {\"count\":2,\"nested\":\"value\"}",
" b_number: 42",
" c_array: [\"first\",3,false]",
" z_string: enabled",
}, "\n")
if !strings.Contains(s, want) {
t.Fatalf("expected deterministic extra_params block %q, got:\n%s", want, s)
}
}
func TestTextFormatterDoesNotIncludeResolvedAPIKeyValue(t *testing.T) { func TestTextFormatterDoesNotIncludeResolvedAPIKeyValue(t *testing.T) {
const secret = "super-secret-api-key" const secret = "super-secret-api-key"
t.Setenv("SCRIPTORIUM_API_KEY", secret) t.Setenv("SCRIPTORIUM_API_KEY", secret)
@@ -61,8 +92,93 @@ func TestTextFormatterDoesNotIncludeResolvedAPIKeyValue(t *testing.T) {
} }
} }
func TestTextFormatterDoesNotIncludeDirectAPIKeyValue(t *testing.T) {
const directKey = "direct-format-key"
// PreparedRun intentionally has no field for direct API keys.
prepared := samplePreparedRun()
out, err := FormatPreparedRun(prepared, PreparedRunFormatText)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if strings.Contains(string(out), directKey) {
t.Fatalf("text output should not include direct api key value: %s", out)
}
}
func TestTextFormatterIncludesMessageCacheControlBeforeContent(t *testing.T) {
prepared := samplePreparedRun()
prepared.Messages = []promptkit.RenderedMessage{
{
Role: "system",
Content: "System guidance.",
CacheControl: &promptkit.CacheControl{
Type: promptkit.CacheControlEphemeral,
TTL: "1h",
},
},
{Role: "user", Content: "Summarize the transcript."},
}
out, err := FormatPreparedRun(prepared, PreparedRunFormatText)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
s := string(out)
if !strings.Contains(s, " system:\n - message: 1\n cache_control: ephemeral ttl=1h\n content: |") {
t.Fatalf("expected system message cache control before content, got:\n%s", s)
}
if strings.Count(s, "cache_control:") != 1 {
t.Fatalf("expected exactly one cache_control line, got:\n%s", s)
}
}
func TestTextFormatterIncludesSessionIDWhenPresent(t *testing.T) {
prepared := samplePreparedRun()
prepared.SessionID = "session-123"
out, err := FormatPreparedRun(prepared, PreparedRunFormatText)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if !strings.Contains(string(out), "session_id: session-123\n") {
t.Fatalf("expected session_id in text output, got:\n%s", out)
}
}
func TestTextFormatterOmitsEmptyCacheControlTTL(t *testing.T) {
prepared := samplePreparedRun()
prepared.Messages = []promptkit.RenderedMessage{
{
Role: "system",
Content: "System guidance.",
CacheControl: &promptkit.CacheControl{
Type: promptkit.CacheControlEphemeral,
},
},
}
out, err := FormatPreparedRun(prepared, PreparedRunFormatText)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
s := string(out)
if !strings.Contains(s, " cache_control: ephemeral\n") {
t.Fatalf("expected cache_control line without ttl, got:\n%s", s)
}
if strings.Contains(s, "ttl=") {
t.Fatalf("expected empty ttl to be omitted, got:\n%s", s)
}
}
func TestJSONFormatterEmitsValidJSONAndIncludesPreparedRunFields(t *testing.T) { func TestJSONFormatterEmitsValidJSONAndIncludesPreparedRunFields(t *testing.T) {
prepared := samplePreparedRun() prepared := samplePreparedRun()
prepared.SessionID = "session-123"
prepared.EffectiveModelParams.ExtraParams = map[string]any{
"number": 42,
"nested": map[string]any{
"enabled": true,
},
}
out, err := FormatPreparedRun(prepared, PreparedRunFormatJSON) out, err := FormatPreparedRun(prepared, PreparedRunFormatJSON)
if err != nil { if err != nil {
@@ -86,9 +202,24 @@ func TestJSONFormatterEmitsValidJSONAndIncludesPreparedRunFields(t *testing.T) {
if decoded["rendered_prompt_hash"] != "rendered-hash" { if decoded["rendered_prompt_hash"] != "rendered-hash" {
t.Fatalf("expected rendered_prompt_hash in json output, got %#v", decoded["rendered_prompt_hash"]) t.Fatalf("expected rendered_prompt_hash in json output, got %#v", decoded["rendered_prompt_hash"])
} }
if _, ok := decoded["effective_model_params"]; !ok { if decoded["session_id"] != "session-123" {
t.Fatalf("expected session_id in json output, got %#v", decoded["session_id"])
}
modelParams, ok := decoded["effective_model_params"].(map[string]any)
if !ok {
t.Fatalf("expected effective_model_params in json output, got %#v", decoded) t.Fatalf("expected effective_model_params in json output, got %#v", decoded)
} }
extraParams, ok := modelParams["extra_params"].(map[string]any)
if !ok {
t.Fatalf("expected extra_params in json output, got %#v", modelParams["extra_params"])
}
if extraParams["number"] != float64(42) {
t.Fatalf("unexpected numeric extra param in json output: %#v", extraParams["number"])
}
nested, ok := extraParams["nested"].(map[string]any)
if !ok || nested["enabled"] != true {
t.Fatalf("unexpected nested extra param in json output: %#v", extraParams["nested"])
}
if _, ok := decoded["input_hashes"]; !ok { if _, ok := decoded["input_hashes"]; !ok {
t.Fatalf("expected input_hashes in json output, got %#v", decoded) t.Fatalf("expected input_hashes in json output, got %#v", decoded)
} }
@@ -97,6 +228,47 @@ func TestJSONFormatterEmitsValidJSONAndIncludesPreparedRunFields(t *testing.T) {
} }
} }
func TestJSONFormatterIncludesMessageCacheControlOnlyWhenPresent(t *testing.T) {
prepared := samplePreparedRun()
prepared.Messages = []promptkit.RenderedMessage{
{
Role: "system",
Content: "System guidance.",
CacheControl: &promptkit.CacheControl{
Type: promptkit.CacheControlEphemeral,
TTL: "1h",
},
},
{Role: "user", Content: "Summarize the transcript."},
}
out, err := FormatPreparedRun(prepared, PreparedRunFormatJSON)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
var decoded struct {
Messages []map[string]any `json:"messages"`
}
if err := json.Unmarshal(out, &decoded); err != nil {
t.Fatalf("expected valid json output, got %v", err)
}
if len(decoded.Messages) != 2 {
t.Fatalf("expected 2 messages, got %d", len(decoded.Messages))
}
cacheControl, ok := decoded.Messages[0]["cache_control"].(map[string]any)
if !ok {
t.Fatalf("expected first message cache_control, got %#v", decoded.Messages[0])
}
if cacheControl["type"] != string(promptkit.CacheControlEphemeral) || cacheControl["ttl"] != "1h" {
t.Fatalf("unexpected cache_control payload: %#v", cacheControl)
}
if _, ok := decoded.Messages[1]["cache_control"]; ok {
t.Fatalf("expected second message to omit cache_control, got %#v", decoded.Messages[1])
}
}
func TestJSONFormatterDoesNotIncludeResolvedAPIKeyValue(t *testing.T) { func TestJSONFormatterDoesNotIncludeResolvedAPIKeyValue(t *testing.T) {
const secret = "super-secret-api-key" const secret = "super-secret-api-key"
t.Setenv("SCRIPTORIUM_API_KEY", secret) t.Setenv("SCRIPTORIUM_API_KEY", secret)
@@ -110,6 +282,19 @@ func TestJSONFormatterDoesNotIncludeResolvedAPIKeyValue(t *testing.T) {
} }
} }
func TestJSONFormatterDoesNotIncludeDirectAPIKeyValue(t *testing.T) {
const directKey = "direct-format-key"
// PreparedRun intentionally has no field for direct API keys.
prepared := samplePreparedRun()
out, err := FormatPreparedRun(prepared, PreparedRunFormatJSON)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if strings.Contains(string(out), directKey) {
t.Fatalf("json output should not include direct api key value: %s", out)
}
}
func TestParsePreparedRunOutputFormatRecognizesSupportedNames(t *testing.T) { func TestParsePreparedRunOutputFormatRecognizesSupportedNames(t *testing.T) {
tests := []struct { tests := []struct {
name string name string
@@ -155,19 +340,20 @@ func TestFormatPreparedRunByNameUnknownFailsClearly(t *testing.T) {
} }
} }
func samplePreparedRun() *domain.PreparedRun { func samplePreparedRun() *promptkit.PreparedRun {
return &domain.PreparedRun{ return &promptkit.PreparedRun{
PromptID: "prompt.id", PromptID: "prompt.id",
PromptVersion: "v1", PromptVersion: "v1",
PromptHash: "prompt-hash", PromptHash: "prompt-hash",
SelectedProfileID: "local-fast", SelectedProfileID: "local-fast",
EffectiveModelParams: domain.ExecutionTarget{ EffectiveModelParams: promptkit.ExecutionTarget{
Endpoint: "http://llm/v1", Endpoint: "http://llm/v1",
Model: "gpt-test", Model: "gpt-test",
Temperature: 0.4, Temperature: 0.4,
MaxTokens: 256, MaxTokens: 256,
TopP: 0.8, TopP: 0.8,
TimeoutSeconds: 45, TimeoutSeconds: 45,
ServiceTier: "priority",
ReasoningEffort: "medium", ReasoningEffort: "medium",
APIKeyEnv: "SCRIPTORIUM_API_KEY", APIKeyEnv: "SCRIPTORIUM_API_KEY",
}, },
@@ -176,7 +362,7 @@ func samplePreparedRun() *domain.PreparedRun {
"glossary": "hash-glossary", "glossary": "hash-glossary",
}, },
RenderedPromptHash: "rendered-hash", RenderedPromptHash: "rendered-hash",
Messages: []domain.RenderedMessage{ Messages: []promptkit.RenderedMessage{
{Role: "system", Content: "System guidance."}, {Role: "system", Content: "System guidance."},
{Role: "user", Content: "Summarize the transcript.\nInclude key entities."}, {Role: "user", Content: "Summarize the transcript.\nInclude key entities."},
{Role: "user", Content: "Second user message."}, {Role: "user", Content: "Second user message."},

View File

@@ -1,11 +0,0 @@
package llm
import (
"context"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
)
// Client executes a rendered prompt against an LLM endpoint.
type Client interface {
Generate(ctx context.Context, req domain.GenerateRequest) (*domain.GenerateResponse, error)
}

View File

@@ -1,253 +0,0 @@
package llm
import (
"bytes"
"context"
"encoding/json"
"errors"
"fmt"
"io"
"net/http"
"net/url"
"os"
"strings"
"time"
"gitea.maximumdirect.net/eric/scriptorium/internal/defaults"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
)
var (
ErrInvalidConfig = errors.New("invalid llm client configuration")
ErrInvalidRequest = errors.New("invalid generate request")
ErrRequestFailed = errors.New("llm request failed")
ErrUnexpectedStatus = errors.New("llm returned non-success status")
ErrMalformedResponse = errors.New("malformed llm response")
)
type OpenAICompatibleConfig struct {
BaseURL string
Model string
Timeout time.Duration
HTTPClient *http.Client
}
type OpenAICompatibleClient struct {
baseURL string
defaultModel string
timeout time.Duration
httpClient *http.Client
}
func NewOpenAICompatibleClient(cfg OpenAICompatibleConfig) (*OpenAICompatibleClient, error) {
baseURL := strings.TrimSpace(cfg.BaseURL)
if baseURL != "" {
if _, err := url.ParseRequestURI(baseURL); err != nil {
return nil, fmt.Errorf("%w: invalid base URL: %v", ErrInvalidConfig, err)
}
}
timeout := cfg.Timeout
if timeout <= 0 {
timeout = defaults.LLMRequestTimeoutDefault
}
var client *http.Client
if cfg.HTTPClient != nil {
client = cfg.HTTPClient
if client.Timeout == 0 {
client.Timeout = timeout
}
} else {
client = &http.Client{Timeout: timeout}
}
return &OpenAICompatibleClient{
baseURL: strings.TrimRight(baseURL, "/"),
defaultModel: cfg.Model,
timeout: timeout,
httpClient: client,
}, nil
}
func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.GenerateRequest) (*domain.GenerateResponse, error) {
if req.Target.TimeoutSeconds < 0 {
return nil, fmt.Errorf("%w: timeout_seconds must be greater than or equal to 0", ErrInvalidRequest)
}
model := strings.TrimSpace(req.Target.Model)
if model == "" {
model = strings.TrimSpace(c.defaultModel)
}
if model == "" {
return nil, fmt.Errorf("%w: model is required", ErrInvalidRequest)
}
endpoint := strings.TrimSpace(req.Target.Endpoint)
if endpoint == "" {
endpoint = c.baseURL
}
if endpoint == "" {
return nil, fmt.Errorf("%w: endpoint is required", ErrInvalidRequest)
}
endpoint = strings.TrimRight(endpoint, "/") + defaults.OpenAIChatCompletionsPath
wireReq := openAIChatRequest{
Model: model,
}
wireReq.Messages = make([]openAIChatMessage, 0, len(req.Prompt.Messages))
for _, msg := range req.Prompt.Messages {
wireReq.Messages = append(wireReq.Messages, openAIChatMessage{
Role: msg.Role,
Content: msg.Content,
})
}
if req.Target.Temperature != 0 {
wireReq.Temperature = &req.Target.Temperature
}
if req.Target.MaxTokens != 0 {
wireReq.MaxTokens = &req.Target.MaxTokens
}
if req.Target.TopP != 0 {
wireReq.TopP = &req.Target.TopP
}
if req.StructuredOutput != nil {
responseFormat, err := toOpenAIResponseFormat(req.StructuredOutput)
if err != nil {
return nil, fmt.Errorf("%w: %v", ErrInvalidRequest, err)
}
wireReq.ResponseFormat = responseFormat
}
payload, err := json.Marshal(wireReq)
if err != nil {
return nil, fmt.Errorf("%w: failed to encode request: %v", ErrRequestFailed, err)
}
httpReq, err := http.NewRequestWithContext(ctx, http.MethodPost, endpoint, bytes.NewReader(payload))
if err != nil {
return nil, fmt.Errorf("%w: failed to create request: %v", ErrRequestFailed, err)
}
httpReq.Header.Set("Content-Type", "application/json")
if envName := strings.TrimSpace(req.Target.APIKeyEnv); envName != "" {
apiKey := strings.TrimSpace(os.Getenv(envName))
if apiKey == "" {
return nil, fmt.Errorf("%w: api key environment variable %q is not set", ErrInvalidRequest, envName)
}
httpReq.Header.Set("Authorization", "Bearer "+apiKey)
}
effectiveTimeout := c.timeout
if req.Target.TimeoutSeconds > 0 {
effectiveTimeout = time.Duration(req.Target.TimeoutSeconds) * time.Second
}
httpClient := c.httpClient
if httpClient == nil {
httpClient = &http.Client{Timeout: effectiveTimeout}
} else if httpClient.Timeout != effectiveTimeout {
cloned := *httpClient
cloned.Timeout = effectiveTimeout
httpClient = &cloned
}
httpResp, err := httpClient.Do(httpReq)
if err != nil {
return nil, fmt.Errorf("%w: %v", ErrRequestFailed, err)
}
defer httpResp.Body.Close()
if httpResp.StatusCode < 200 || httpResp.StatusCode >= 300 {
body, _ := io.ReadAll(io.LimitReader(httpResp.Body, 4096))
return nil, fmt.Errorf("%w: status=%d body=%q", ErrUnexpectedStatus, httpResp.StatusCode, strings.TrimSpace(string(body)))
}
var wireResp openAIChatResponse
if err := json.NewDecoder(httpResp.Body).Decode(&wireResp); err != nil {
return nil, fmt.Errorf("%w: failed to decode response: %v", ErrMalformedResponse, err)
}
if len(wireResp.Choices) == 0 {
return nil, fmt.Errorf("%w: no choices returned", ErrMalformedResponse)
}
content := wireResp.Choices[0].Message.Content
if content == "" {
return nil, fmt.Errorf("%w: first choice has empty message content", ErrMalformedResponse)
}
return &domain.GenerateResponse{
Content: content,
Usage: domain.TokenUsage{
PromptTokens: wireResp.Usage.PromptTokens,
CompletionTokens: wireResp.Usage.CompletionTokens,
TotalTokens: wireResp.Usage.TotalTokens,
},
}, nil
}
type openAIChatRequest struct {
Model string `json:"model"`
Messages []openAIChatMessage `json:"messages"`
Temperature *float64 `json:"temperature,omitempty"`
MaxTokens *int `json:"max_tokens,omitempty"`
TopP *float64 `json:"top_p,omitempty"`
ResponseFormat *openAIResponseFormat `json:"response_format,omitempty"`
}
type openAIChatMessage struct {
Role string `json:"role"`
Content string `json:"content"`
}
type openAIChatResponse struct {
Choices []struct {
Message openAIChatMessage `json:"message"`
} `json:"choices"`
Usage struct {
PromptTokens int `json:"prompt_tokens"`
CompletionTokens int `json:"completion_tokens"`
TotalTokens int `json:"total_tokens"`
} `json:"usage"`
}
type openAIResponseFormat struct {
Type string `json:"type"`
JSONSchema *openAIJSONSchemaEnvelope `json:"json_schema,omitempty"`
}
type openAIJSONSchemaEnvelope struct {
Name string `json:"name"`
Strict bool `json:"strict"`
Schema any `json:"schema"`
}
func toOpenAIResponseFormat(spec *domain.StructuredOutputSpec) (*openAIResponseFormat, error) {
if spec == nil {
return nil, nil
}
switch spec.Type {
case domain.StructuredOutputJSONSchema:
if spec.JSONSchema == nil {
return nil, errors.New("json_schema structured output requires schema payload")
}
if strings.TrimSpace(spec.JSONSchema.Name) == "" {
return nil, errors.New("json_schema structured output requires non-empty schema name")
}
if spec.JSONSchema.Schema == nil {
return nil, errors.New("json_schema structured output requires schema document")
}
return &openAIResponseFormat{
Type: "json_schema",
JSONSchema: &openAIJSONSchemaEnvelope{
Name: spec.JSONSchema.Name,
Strict: spec.JSONSchema.Strict,
Schema: spec.JSONSchema.Schema,
},
}, nil
default:
return nil, fmt.Errorf("unsupported structured output type %q", spec.Type)
}
}

View File

@@ -1,467 +0,0 @@
package llm
import (
"context"
"encoding/json"
"errors"
"net/http"
"net/http/httptest"
"strings"
"testing"
"time"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
)
func TestOpenAICompatibleClientGenerateSuccess(t *testing.T) {
type observedRequest struct {
Authorization string
Body map[string]any
}
obs := &observedRequest{}
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
obs.Authorization = r.Header.Get("Authorization")
if r.URL.Path != "/v1/chat/completions" {
t.Fatalf("unexpected path: %s", r.URL.Path)
}
if ct := r.Header.Get("Content-Type"); ct != "application/json" {
t.Fatalf("unexpected content type: %s", ct)
}
defer r.Body.Close()
if err := json.NewDecoder(r.Body).Decode(&obs.Body); err != nil {
t.Fatalf("failed to decode request body: %v", err)
}
w.Header().Set("Content-Type", "application/json")
_, _ = w.Write([]byte(`{
"choices": [{"message": {"role": "assistant", "content": "hello from model"}}],
"usage": {"prompt_tokens": 11, "completion_tokens": 22, "total_tokens": 33}
}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{
BaseURL: ts.URL + "/v1",
Timeout: 2 * time.Second,
})
if err != nil {
t.Fatalf("unexpected constructor error: %v", err)
}
t.Setenv("SCRIPTORIUM_TEST_API_KEY", "secret-key")
resp, err := client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{
{Role: "system", Content: "You are helpful."},
{Role: "user", Content: "Say hello"},
}},
Target: domain.ExecutionTarget{
Model: "gpt-test",
Temperature: 0.4,
MaxTokens: 123,
TopP: 0.7,
APIKeyEnv: "SCRIPTORIUM_TEST_API_KEY",
},
StructuredOutput: &domain.StructuredOutputSpec{
Type: domain.StructuredOutputJSONSchema,
JSONSchema: &domain.StructuredOutputJSONSpec{
Name: "weather_schema",
Strict: true,
Schema: map[string]any{
"type": "object",
"properties": map[string]any{
"location": map[string]any{"type": "string"},
},
"required": []any{"location"},
},
},
},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if resp.Content != "hello from model" {
t.Fatalf("unexpected content: %q", resp.Content)
}
if resp.Usage.PromptTokens != 11 || resp.Usage.CompletionTokens != 22 || resp.Usage.TotalTokens != 33 {
t.Fatalf("unexpected usage: %+v", resp.Usage)
}
if obs.Authorization != "Bearer secret-key" {
t.Fatalf("unexpected Authorization header: %q", obs.Authorization)
}
if got, ok := obs.Body["model"].(string); !ok || got != "gpt-test" {
t.Fatalf("unexpected model payload: %#v", obs.Body["model"])
}
msgs, ok := obs.Body["messages"].([]any)
if !ok || len(msgs) != 2 {
t.Fatalf("unexpected messages payload: %#v", obs.Body["messages"])
}
msg0 := msgs[0].(map[string]any)
if msg0["role"] != "system" || msg0["content"] != "You are helpful." {
t.Fatalf("unexpected first message: %#v", msg0)
}
msg1 := msgs[1].(map[string]any)
if msg1["role"] != "user" || msg1["content"] != "Say hello" {
t.Fatalf("unexpected second message: %#v", msg1)
}
responseFormat, ok := obs.Body["response_format"].(map[string]any)
if !ok {
t.Fatalf("expected response_format payload, got %#v", obs.Body["response_format"])
}
if responseFormat["type"] != "json_schema" {
t.Fatalf("expected response_format.type=json_schema, got %#v", responseFormat["type"])
}
jsonSchema, ok := responseFormat["json_schema"].(map[string]any)
if !ok {
t.Fatalf("expected response_format.json_schema map, got %#v", responseFormat["json_schema"])
}
if jsonSchema["name"] != "weather_schema" {
t.Fatalf("expected json_schema.name weather_schema, got %#v", jsonSchema["name"])
}
if jsonSchema["strict"] != true {
t.Fatalf("expected json_schema.strict=true, got %#v", jsonSchema["strict"])
}
if _, ok := jsonSchema["schema"].(map[string]any); !ok {
t.Fatalf("expected json_schema.schema object, got %#v", jsonSchema["schema"])
}
}
func TestOpenAICompatibleClientOmitsResponseFormatWhenNoStructuredOutput(t *testing.T) {
var observedBody map[string]any
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
defer r.Body.Close()
if err := json.NewDecoder(r.Body).Decode(&observedBody); err != nil {
t.Fatalf("failed to decode request body: %v", err)
}
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{Model: "model"},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if _, exists := observedBody["response_format"]; exists {
t.Fatalf("expected response_format omitted, got %#v", observedBody["response_format"])
}
}
func TestOpenAICompatibleClientNoAuthorizationHeaderWhenNoAPIKey(t *testing.T) {
hadAuth := false
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
hadAuth = r.Header.Get("Authorization") != ""
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{Model: "model"},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if hadAuth {
t.Fatal("did not expect Authorization header")
}
}
func TestOpenAICompatibleClientAPIKeyEnvMissing(t *testing.T) {
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1", Model: "model"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{APIKeyEnv: "SCRIPTORIUM_MISSING_KEY"},
})
if err == nil {
t.Fatal("expected missing API key env error")
}
if !errors.Is(err, ErrInvalidRequest) {
t.Fatalf("expected ErrInvalidRequest, got %v", err)
}
}
func TestOpenAICompatibleClientModelFallbackFromConfig(t *testing.T) {
gotModel := ""
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
var body map[string]any
_ = json.NewDecoder(r.Body).Decode(&body)
if m, ok := body["model"].(string); ok {
gotModel = m
}
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1", Model: "default-model"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if gotModel != "default-model" {
t.Fatalf("expected default model, got %q", gotModel)
}
}
func TestOpenAICompatibleClientEndpointOverride(t *testing.T) {
defaultHit := false
overrideHit := false
defaultServer := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
defaultHit = true
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"default"}}]}`))
}))
defer defaultServer.Close()
overrideServer := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
overrideHit = true
if r.URL.Path != "/v1/chat/completions" {
t.Fatalf("unexpected path: %s", r.URL.Path)
}
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"override"}}]}`))
}))
defer overrideServer.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: defaultServer.URL + "/v1", Model: "m"})
if err != nil {
t.Fatal(err)
}
resp, err := client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{Endpoint: overrideServer.URL + "/v1"},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if resp.Content != "override" {
t.Fatalf("expected override response, got %q", resp.Content)
}
if defaultHit {
t.Fatal("default endpoint should not have been called")
}
if !overrideHit {
t.Fatal("override endpoint should have been called")
}
}
func TestOpenAICompatibleClientNon2xxError(t *testing.T) {
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.WriteHeader(http.StatusBadRequest)
_, _ = w.Write([]byte(`{"error":"bad request payload"}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1", Model: "m"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
})
if err == nil {
t.Fatal("expected non-2xx error")
}
if !errors.Is(err, ErrUnexpectedStatus) {
t.Fatalf("expected ErrUnexpectedStatus, got %v", err)
}
if !strings.Contains(err.Error(), "400") || !strings.Contains(err.Error(), "bad request payload") {
t.Fatalf("expected status/body details, got %v", err)
}
}
func TestOpenAICompatibleClientMalformedResponseInvalidJSON(t *testing.T) {
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
_, _ = w.Write([]byte(`{not valid json`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1", Model: "m"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
})
if err == nil {
t.Fatal("expected malformed response error")
}
if !errors.Is(err, ErrMalformedResponse) {
t.Fatalf("expected ErrMalformedResponse, got %v", err)
}
}
func TestOpenAICompatibleClientMalformedResponseMissingChoices(t *testing.T) {
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
_, _ = w.Write([]byte(`{"choices": []}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1", Model: "m"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
})
if err == nil {
t.Fatal("expected malformed response error")
}
if !errors.Is(err, ErrMalformedResponse) {
t.Fatalf("expected ErrMalformedResponse, got %v", err)
}
}
func TestOpenAICompatibleClientTimeout(t *testing.T) {
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
time.Sleep(250 * time.Millisecond)
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{
BaseURL: ts.URL + "/v1",
Model: "m",
Timeout: 50 * time.Millisecond,
})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
})
if err == nil {
t.Fatal("expected timeout error")
}
if !errors.Is(err, ErrRequestFailed) {
t.Fatalf("expected ErrRequestFailed, got %v", err)
}
}
func TestOpenAICompatibleClientRequestTimeoutOverride(t *testing.T) {
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
time.Sleep(100 * time.Millisecond)
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{
BaseURL: ts.URL + "/v1",
Model: "m",
Timeout: 50 * time.Millisecond,
})
if err != nil {
t.Fatal(err)
}
resp, err := client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{TimeoutSeconds: 1},
})
if err != nil {
t.Fatalf("expected request-level timeout override to succeed, got %v", err)
}
if resp.Content != "ok" {
t.Fatalf("expected response content ok, got %q", resp.Content)
}
}
func TestOpenAICompatibleClientNegativeTimeoutRejected(t *testing.T) {
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{
BaseURL: "http://example.com/v1",
Model: "m",
})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{TimeoutSeconds: -1},
})
if err == nil {
t.Fatal("expected invalid request error")
}
if !errors.Is(err, ErrInvalidRequest) {
t.Fatalf("expected ErrInvalidRequest, got %v", err)
}
}
func TestOpenAICompatibleClientAllowsEmptyConfiguredBaseURL(t *testing.T) {
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{
BaseURL: "",
Model: "m",
})
if err != nil {
t.Fatalf("expected empty configured base URL to be allowed, got %v", err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{Endpoint: "http://localhost:9999/v1"},
})
if err == nil {
t.Fatal("expected request failure due to unreachable endpoint")
}
if !errors.Is(err, ErrRequestFailed) {
t.Fatalf("expected ErrRequestFailed with request endpoint override, got %v", err)
}
}
func TestOpenAICompatibleClientRequiresEndpointWhenUnsetEverywhere(t *testing.T) {
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{
BaseURL: "",
Model: "m",
})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{},
})
if err == nil {
t.Fatal("expected endpoint-required error")
}
if !errors.Is(err, ErrInvalidRequest) {
t.Fatalf("expected ErrInvalidRequest, got %v", err)
}
}

View File

@@ -1,114 +0,0 @@
package profile
import (
"bytes"
"context"
"errors"
"fmt"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
"gopkg.in/yaml.v3"
)
var (
ErrProfileNotFound = errors.New("execution profile not found")
ErrInvalidYAML = errors.New("invalid YAML format")
ErrInvalidProfile = errors.New("invalid execution profile configuration")
ErrRawAPIKeyNotAllowed = errors.New("raw api_key is not allowed; use api_key_env")
)
type filesystemRepository struct {
dir string
}
func NewFilesystemRepository(dir string) Repository {
return &filesystemRepository{dir: dir}
}
func (r *filesystemRepository) GetProfile(ctx context.Context, id string) (*domain.ExecutionProfile, error) {
if strings.TrimSpace(id) == "" {
return nil, fmt.Errorf("%w: profile id is required", ErrInvalidProfile)
}
files, err := os.ReadDir(r.dir)
if err != nil {
return nil, fmt.Errorf("failed to read profile directory: %w", err)
}
for _, file := range files {
select {
case <-ctx.Done():
return nil, ctx.Err()
default:
}
if file.IsDir() || (!strings.HasSuffix(file.Name(), ".yaml") && !strings.HasSuffix(file.Name(), ".yml")) {
continue
}
fullPath := filepath.Join(r.dir, file.Name())
data, err := os.ReadFile(fullPath)
if err != nil {
return nil, fmt.Errorf("failed to read profile file %s: %w", file.Name(), err)
}
var prof domain.ExecutionProfile
decoder := yaml.NewDecoder(bytes.NewReader(data))
decoder.KnownFields(true)
if err := decoder.Decode(&prof); err != nil {
if strings.Contains(err.Error(), "field api_key not found") {
if strings.TrimSuffix(strings.TrimSuffix(file.Name(), ".yaml"), ".yml") == id {
return nil, fmt.Errorf("%w: %s", ErrRawAPIKeyNotAllowed, file.Name())
}
continue
}
if strings.TrimSuffix(strings.TrimSuffix(file.Name(), ".yaml"), ".yml") == id {
return nil, fmt.Errorf("%w: %s: %v", ErrInvalidYAML, file.Name(), err)
}
continue
}
if prof.ID != id {
continue
}
if err := validateProfile(&prof); err != nil {
if errors.Is(err, ErrRawAPIKeyNotAllowed) {
return nil, fmt.Errorf("%w: %s", err, file.Name())
}
return nil, fmt.Errorf("%w: %s: %v", ErrInvalidProfile, file.Name(), err)
}
return &prof, nil
}
return nil, ErrProfileNotFound
}
func validateProfile(p *domain.ExecutionProfile) error {
if strings.TrimSpace(p.ID) == "" {
return errors.New("id is required")
}
if strings.TrimSpace(p.Endpoint) == "" {
return errors.New("endpoint is required")
}
if strings.TrimSpace(p.Model) == "" {
return errors.New("model is required")
}
if p.Temperature < 0 || p.Temperature > 2 {
return errors.New("temperature must be between 0 and 2")
}
if p.MaxTokens < 0 {
return errors.New("max_tokens must be greater than or equal to 0")
}
if p.TopP < 0 || p.TopP > 1 {
return errors.New("top_p must be between 0 and 1")
}
if p.TimeoutSeconds < 0 {
return errors.New("timeout_seconds must be greater than or equal to 0")
}
return nil
}

View File

@@ -1,12 +0,0 @@
package profile
import (
"context"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
)
// Repository loads execution profiles.
type Repository interface {
GetProfile(ctx context.Context, id string) (*domain.ExecutionProfile, error)
}

View File

@@ -1,111 +0,0 @@
package profile
import (
"context"
"errors"
"os"
"path/filepath"
"testing"
)
func TestFilesystemRepository_GetProfile(t *testing.T) {
tmpDir, err := os.MkdirTemp("", "execution_profile_test")
if err != nil {
t.Fatal(err)
}
defer os.RemoveAll(tmpDir)
files, err := os.ReadDir("testdata")
if err != nil {
t.Fatalf("failed to read testdata: %v", err)
}
for _, f := range files {
src := filepath.Join("testdata", f.Name())
dst := filepath.Join(tmpDir, f.Name())
data, err := os.ReadFile(src)
if err != nil {
t.Fatal(err)
}
if err := os.WriteFile(dst, data, 0644); err != nil {
t.Fatal(err)
}
}
repo := NewFilesystemRepository(tmpDir)
ctx := context.Background()
t.Run("valid local profile", func(t *testing.T) {
p, err := repo.GetProfile(ctx, "local-default")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if p.ID != "local-default" {
t.Fatalf("unexpected id: %q", p.ID)
}
if p.Endpoint == "" || p.Model == "" {
t.Fatalf("expected endpoint/model to be set: %+v", p)
}
})
t.Run("valid profile with api_key_env", func(t *testing.T) {
p, err := repo.GetProfile(ctx, "local-secure")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if p.APIKeyEnv != "SCRIPTORIUM_API_KEY" {
t.Fatalf("unexpected api_key_env: %q", p.APIKeyEnv)
}
if p.ReasoningEffort != "medium" {
t.Fatalf("unexpected reasoning_effort: %q", p.ReasoningEffort)
}
})
t.Run("invalid yaml", func(t *testing.T) {
_, err := repo.GetProfile(ctx, "invalid_yaml")
if !errors.Is(err, ErrInvalidYAML) {
t.Fatalf("expected ErrInvalidYAML, got %v", err)
}
})
t.Run("missing id", func(t *testing.T) {
_, err := repo.GetProfile(ctx, "missing_id")
if !errors.Is(err, ErrProfileNotFound) {
t.Fatalf("expected ErrProfileNotFound, got %v", err)
}
})
t.Run("missing endpoint", func(t *testing.T) {
_, err := repo.GetProfile(ctx, "missing-endpoint")
if !errors.Is(err, ErrInvalidProfile) {
t.Fatalf("expected ErrInvalidProfile, got %v", err)
}
})
t.Run("missing model", func(t *testing.T) {
_, err := repo.GetProfile(ctx, "missing-model")
if !errors.Is(err, ErrInvalidProfile) {
t.Fatalf("expected ErrInvalidProfile, got %v", err)
}
})
t.Run("unknown field", func(t *testing.T) {
_, err := repo.GetProfile(ctx, "unknown_field")
if !errors.Is(err, ErrInvalidYAML) {
t.Fatalf("expected ErrInvalidYAML for strict decode unknown field, got %v", err)
}
})
t.Run("raw api_key rejected", func(t *testing.T) {
_, err := repo.GetProfile(ctx, "raw_api_key")
if !errors.Is(err, ErrRawAPIKeyNotAllowed) {
t.Fatalf("expected ErrRawAPIKeyNotAllowed, got %v", err)
}
})
t.Run("profile not found", func(t *testing.T) {
_, err := repo.GetProfile(ctx, "does-not-exist")
if !errors.Is(err, ErrProfileNotFound) {
t.Fatalf("expected ErrProfileNotFound, got %v", err)
}
})
}

View File

@@ -1,3 +0,0 @@
id: invalid_yaml
endpoint: http://localhost:8000/v1
model: [broken

View File

@@ -1,2 +0,0 @@
id: missing-endpoint
model: gpt-4o-mini

View File

@@ -1,2 +0,0 @@
endpoint: http://localhost:8000/v1
model: gpt-4o-mini

View File

@@ -1,2 +0,0 @@
id: missing-model
endpoint: http://localhost:8000/v1

View File

@@ -1,4 +0,0 @@
id: raw-api-key
endpoint: http://localhost:8000/v1
model: gpt-4o-mini
api_key: super-secret-should-not-be-here

View File

@@ -1,4 +0,0 @@
id: unknown-field
endpoint: http://localhost:8000/v1
model: gpt-4o-mini
foo: bar

View File

@@ -1,7 +0,0 @@
id: local-default
endpoint: http://localhost:8000/v1
model: gpt-4o-mini
temperature: 0.2
max_tokens: 700
top_p: 1.0
timeout_seconds: 120

View File

@@ -1,7 +0,0 @@
id: local-secure
endpoint: http://localhost:8000/v1
model: gpt-4o-mini
api_key_env: SCRIPTORIUM_API_KEY
reasoning_effort: medium
extra_params:
provider: local

View File

@@ -1,86 +0,0 @@
package prompt
import (
"bytes"
"context"
"errors"
"fmt"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
"text/template"
)
var (
ErrMissingRequiredInput = errors.New("missing required input artifact")
ErrUnknownInput = errors.New("referenced unknown input artifact")
ErrInvalidTemplate = errors.New("invalid prompt template")
ErrRenderFailure = errors.New("prompt render failure")
ErrInvalidMessageRole = errors.New("invalid or empty message role")
)
type goRenderer struct{}
func NewGoRenderer() Renderer {
return &goRenderer{}
}
func (r *goRenderer) Render(ctx context.Context, definition *domain.PromptDefinition, inputs map[string]*domain.Artifact, vars map[string]string) (*domain.RenderedPrompt, error) {
if definition == nil {
return nil, fmt.Errorf("%w: nil prompt definition", ErrRenderFailure)
}
// 1. Verify required inputs
for _, in := range definition.Inputs {
if !in.Required {
continue
}
art, ok := inputs[in.Name]
if !ok || art == nil {
return nil, fmt.Errorf("%w: %s", ErrMissingRequiredInput, in.Name)
}
}
// 2. Setup template functions
funcs := template.FuncMap{
"input": func(name string) (string, error) {
art, ok := inputs[name]
if !ok || art == nil {
return "", fmt.Errorf("%w: %s", ErrUnknownInput, name)
}
return string(art.Body), nil
},
}
var renderedMessages []domain.RenderedMessage
for i, tmplMsg := range definition.Templates {
select {
case <-ctx.Done():
return nil, ctx.Err()
default:
}
if tmplMsg.Role == "" {
return nil, fmt.Errorf("%w: message %d", ErrInvalidMessageRole, i)
}
// Parse and execute template
tmpl, err := template.New(fmt.Sprintf("msg_%d", i)).Funcs(funcs).Option("missingkey=error").Parse(tmplMsg.Content)
if err != nil {
return nil, fmt.Errorf("%w: message %d: %v", ErrInvalidTemplate, i, err)
}
var buf bytes.Buffer
if err := tmpl.Execute(&buf, vars); err != nil {
return nil, fmt.Errorf("%w: message %d: %w", ErrRenderFailure, i, err)
}
renderedMessages = append(renderedMessages, domain.RenderedMessage{
Role: tmplMsg.Role,
Content: buf.String(),
})
}
return &domain.RenderedPrompt{
Messages: renderedMessages,
}, nil
}

View File

@@ -1,11 +0,0 @@
package prompt
import (
"context"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
)
// Renderer renders prompt templates using named artifacts and variables.
type Renderer interface {
Render(ctx context.Context, definition *domain.PromptDefinition, inputs map[string]*domain.Artifact, vars map[string]string) (*domain.RenderedPrompt, error)
}

View File

@@ -1,212 +0,0 @@
package prompt
import (
"context"
"errors"
"testing"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
)
func TestGoRenderer_Render(t *testing.T) {
renderer := NewGoRenderer()
ctx := context.Background()
inputs := map[string]*domain.Artifact{
"transcript": {Body: []byte("The quick brown fox.")},
}
vars := map[string]string{
"role": "helpful assistant",
"tone": "concise",
}
t.Run("rendering inline message content", func(t *testing.T) {
def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "user", Content: "Analyze this: {{input \"transcript\"}}"},
},
}
res, err := renderer.Render(ctx, def, inputs, vars)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if len(res.Messages) != 1 {
t.Fatalf("expected 1 message, got %d", len(res.Messages))
}
if res.Messages[0].Content != "Analyze this: The quick brown fox." {
t.Fatalf("unexpected rendered content: %q", res.Messages[0].Content)
}
})
t.Run("rendering file-backed message content loaded into prompt definition", func(t *testing.T) {
def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "user", Content: "From file: {{input \"transcript\"}}", ContentFile: "/tmp/user.tmpl"},
},
}
res, err := renderer.Render(ctx, def, inputs, vars)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if got := res.Messages[0].Content; got != "From file: The quick brown fox." {
t.Fatalf("unexpected file-backed render result: %q", got)
}
})
t.Run("rendering system and user messages", func(t *testing.T) {
def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "system", Content: "You are a {{.role}}."},
{Role: "user", Content: "Analyze this: {{input \"transcript\"}}"},
},
}
res, err := renderer.Render(ctx, def, inputs, vars)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if len(res.Messages) != 2 {
t.Fatalf("expected 2 messages, got %d", len(res.Messages))
}
if res.Messages[0].Role != "system" || res.Messages[1].Role != "user" {
t.Fatalf("unexpected roles: %#v", res.Messages)
}
})
t.Run("accessing vars", func(t *testing.T) {
def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "system", Content: "Speak in a {{.tone}} tone."},
},
}
res, err := renderer.Render(ctx, def, inputs, vars)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if res.Messages[0].Content != "Speak in a concise tone." {
t.Fatalf("unexpected vars rendering: %q", res.Messages[0].Content)
}
})
t.Run("inserting required input artifact", func(t *testing.T) {
def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "user", Content: "{{input \"transcript\"}}"},
},
}
res, err := renderer.Render(ctx, def, inputs, vars)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if res.Messages[0].Content != "The quick brown fox." {
t.Fatalf("unexpected required input rendering: %q", res.Messages[0].Content)
}
})
t.Run("optional input absent and not referenced", func(t *testing.T) {
def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{
{Name: "transcript", Required: true},
{Name: "glossary", Required: false},
},
Templates: []domain.PromptMessageTemplate{
{Role: "user", Content: "Transcript: {{input \"transcript\"}}"},
},
}
res, err := renderer.Render(ctx, def, inputs, vars)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if len(res.Messages) != 1 {
t.Fatalf("expected one rendered message, got %d", len(res.Messages))
}
})
t.Run("optional input absent but referenced, expecting failure", func(t *testing.T) {
def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{
{Name: "transcript", Required: true},
{Name: "glossary", Required: false},
},
Templates: []domain.PromptMessageTemplate{
{Role: "user", Content: "Glossary: {{input \"glossary\"}}"},
},
}
_, err := renderer.Render(ctx, def, inputs, vars)
if !errors.Is(err, ErrRenderFailure) {
t.Fatalf("expected ErrRenderFailure, got %v", err)
}
if !errors.Is(err, ErrUnknownInput) {
t.Fatalf("expected ErrUnknownInput, got %v", err)
}
})
t.Run("required input missing, expecting failure", func(t *testing.T) {
def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "user", Content: "Analyze this: {{input \"transcript\"}}"},
},
}
_, err := renderer.Render(ctx, def, map[string]*domain.Artifact{}, vars)
if !errors.Is(err, ErrMissingRequiredInput) {
t.Fatalf("expected ErrMissingRequiredInput, got %v", err)
}
})
t.Run("invalid template syntax", func(t *testing.T) {
def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "user", Content: "Hello {{.unclosed"},
},
}
_, err := renderer.Render(ctx, def, inputs, vars)
if !errors.Is(err, ErrInvalidTemplate) {
t.Fatalf("expected ErrInvalidTemplate, got %v", err)
}
})
t.Run("unknown input reference", func(t *testing.T) {
def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "user", Content: "Hello {{input \"ghost\"}}"},
},
}
_, err := renderer.Render(ctx, def, inputs, vars)
if !errors.Is(err, ErrRenderFailure) {
t.Fatalf("expected ErrRenderFailure, got %v", err)
}
if !errors.Is(err, ErrUnknownInput) {
t.Fatalf("expected ErrUnknownInput, got %v", err)
}
})
t.Run("empty message role", func(t *testing.T) {
def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "", Content: "Hello"},
},
}
_, err := renderer.Render(ctx, def, inputs, vars)
if !errors.Is(err, ErrInvalidMessageRole) {
t.Fatalf("expected ErrInvalidMessageRole, got %v", err)
}
})
}

View File

@@ -1,268 +0,0 @@
package promptdef
import (
"bytes"
"context"
"errors"
"fmt"
"os"
"path/filepath"
"strings"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
"gopkg.in/yaml.v3"
)
var (
ErrPromptDefinitionNotFound = errors.New("prompt definition not found")
ErrInvalidYAML = errors.New("invalid YAML format")
ErrInvalidPromptDefinition = errors.New("invalid prompt definition configuration")
)
type filesystemRepository struct {
dir string
}
type promptDefinitionFile struct {
ID string `yaml:"id"`
Version string `yaml:"version"`
DefaultProfile *string `yaml:"default_profile"`
Description string `yaml:"description"`
Inputs []promptInputFile `yaml:"inputs"`
Messages []promptMessageFile `yaml:"messages"`
Output promptOutputContractFile `yaml:"output"`
}
type promptInputFile struct {
Name string `yaml:"name"`
Required bool `yaml:"required"`
ContentType string `yaml:"content_type"`
Description string `yaml:"description"`
}
type promptMessageFile struct {
Role string `yaml:"role"`
Content string `yaml:"content"`
ContentFile string `yaml:"content_file"`
}
type promptOutputContractFile struct {
Format domain.OutputFormat `yaml:"format"`
ValidationMode domain.ValidationMode `yaml:"validation_mode"`
SchemaPath string `yaml:"schema_path"`
RepairAttempts int `yaml:"repair_attempts"`
}
func NewFilesystemRepository(dir string) Repository {
return &filesystemRepository{dir: dir}
}
func (r *filesystemRepository) GetPromptDefinition(ctx context.Context, id string, version string) (*domain.PromptDefinition, error) {
if strings.TrimSpace(id) == "" {
return nil, fmt.Errorf("%w: prompt id is required", ErrInvalidPromptDefinition)
}
files, err := os.ReadDir(r.dir)
if err != nil {
return nil, fmt.Errorf("failed to read prompt definition directory: %w", err)
}
for _, file := range files {
select {
case <-ctx.Done():
return nil, ctx.Err()
default:
}
if file.IsDir() || !isYAMLFile(file.Name()) {
continue
}
fullPath := filepath.Join(r.dir, file.Name())
fileMatch := promptIDFromFileName(file.Name()) == id
raw, err := loadPromptDefinitionFile(fullPath)
if err != nil {
if fileMatch {
return nil, fmt.Errorf("%w: %s: %v", ErrInvalidYAML, file.Name(), err)
}
continue
}
def, err := normalizePromptDefinition(raw, fullPath)
if err != nil {
if fileMatch || strings.TrimSpace(raw.ID) == id {
return nil, fmt.Errorf("%w: %s: %v", ErrInvalidPromptDefinition, file.Name(), err)
}
continue
}
if def.ID != id {
continue
}
if version != "" && def.Version != version {
continue
}
return def, nil
}
return nil, ErrPromptDefinitionNotFound
}
func loadPromptDefinitionFile(path string) (*promptDefinitionFile, error) {
data, err := os.ReadFile(path)
if err != nil {
return nil, fmt.Errorf("failed to read prompt definition file: %w", err)
}
var raw promptDefinitionFile
decoder := yaml.NewDecoder(bytes.NewReader(data))
decoder.KnownFields(true)
if err := decoder.Decode(&raw); err != nil {
return nil, err
}
return &raw, nil
}
func normalizePromptDefinition(raw *promptDefinitionFile, sourcePath string) (*domain.PromptDefinition, error) {
if raw == nil {
return nil, errors.New("prompt definition is nil")
}
id := strings.TrimSpace(raw.ID)
if id == "" {
return nil, errors.New("id is required")
}
version := strings.TrimSpace(raw.Version)
if version == "" {
return nil, errors.New("version is required")
}
if len(raw.Messages) == 0 {
return nil, errors.New("at least one message is required")
}
inputs := make([]domain.PromptInput, 0, len(raw.Inputs))
seenInputNames := make(map[string]struct{}, len(raw.Inputs))
for i, in := range raw.Inputs {
name := strings.TrimSpace(in.Name)
if name == "" {
return nil, fmt.Errorf("input %d has empty name", i)
}
if _, exists := seenInputNames[name]; exists {
return nil, fmt.Errorf("duplicate input name %q", name)
}
seenInputNames[name] = struct{}{}
inputs = append(inputs, domain.PromptInput{
Name: name,
Required: in.Required,
ContentType: strings.TrimSpace(in.ContentType),
Description: strings.TrimSpace(in.Description),
})
}
templates := make([]domain.PromptMessageTemplate, 0, len(raw.Messages))
promptDir := filepath.Dir(sourcePath)
for i, msg := range raw.Messages {
role := strings.TrimSpace(msg.Role)
if role == "" {
return nil, fmt.Errorf("message %d role is required", i)
}
hasContent := strings.TrimSpace(msg.Content) != ""
hasContentFile := strings.TrimSpace(msg.ContentFile) != ""
if hasContent == hasContentFile {
return nil, fmt.Errorf("message %d (%s) must set exactly one of content or content_file", i, role)
}
templateContent := msg.Content
resolvedContentFile := ""
if hasContentFile {
resolvedPath := strings.TrimSpace(msg.ContentFile)
if !filepath.IsAbs(resolvedPath) {
resolvedPath = filepath.Join(promptDir, resolvedPath)
}
resolvedPath = filepath.Clean(resolvedPath)
body, err := os.ReadFile(resolvedPath)
if err != nil {
return nil, fmt.Errorf("prompt %q message %d (%s): failed to read content_file %q: %w", id, i, role, msg.ContentFile, err)
}
templateContent = string(body)
resolvedContentFile = resolvedPath
}
templates = append(templates, domain.PromptMessageTemplate{
Role: role,
Content: templateContent,
ContentFile: resolvedContentFile,
})
}
if !isValidOutputFormat(raw.Output.Format) {
return nil, fmt.Errorf("invalid output format: %q", raw.Output.Format)
}
if !isValidValidationMode(raw.Output.ValidationMode) {
return nil, fmt.Errorf("invalid validation mode: %q", raw.Output.ValidationMode)
}
if raw.Output.ValidationMode == domain.ValidationJSONSchema && strings.TrimSpace(raw.Output.SchemaPath) == "" {
return nil, errors.New("output.schema_path is required when output.validation_mode is json_schema")
}
if raw.Output.RepairAttempts < 0 {
return nil, errors.New("output.repair_attempts must be greater than or equal to 0")
}
defaultProfile := ""
if raw.DefaultProfile != nil {
defaultProfile = strings.TrimSpace(*raw.DefaultProfile)
if defaultProfile == "" {
return nil, errors.New("default_profile must be a non-empty string when set")
}
}
return &domain.PromptDefinition{
ID: id,
Version: version,
DefaultProfile: defaultProfile,
Description: strings.TrimSpace(raw.Description),
Inputs: inputs,
Templates: templates,
OutputFormat: raw.Output.Format,
Validation: domain.OutputContract{
Format: raw.Output.Format,
ValidationMode: raw.Output.ValidationMode,
SchemaPath: strings.TrimSpace(raw.Output.SchemaPath),
RepairAttempts: raw.Output.RepairAttempts,
},
}, nil
}
func isYAMLFile(name string) bool {
return strings.HasSuffix(name, ".yaml") || strings.HasSuffix(name, ".yml")
}
func promptIDFromFileName(name string) string {
name = strings.TrimSuffix(name, ".yaml")
name = strings.TrimSuffix(name, ".yml")
return name
}
func isValidOutputFormat(f domain.OutputFormat) bool {
switch f {
case domain.FormatText, domain.FormatMarkdown, domain.FormatJSON:
return true
default:
return false
}
}
func isValidValidationMode(m domain.ValidationMode) bool {
switch m {
case domain.ValidationNone, domain.ValidationBasic, domain.ValidationJSON, domain.ValidationJSONSchema:
return true
default:
return false
}
}

View File

@@ -1,12 +0,0 @@
package promptdef
import (
"context"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
)
// Repository loads prompt definitions.
type Repository interface {
GetPromptDefinition(ctx context.Context, id string, version string) (*domain.PromptDefinition, error)
}

View File

@@ -1,158 +0,0 @@
package promptdef
import (
"context"
"errors"
"io/fs"
"os"
"path/filepath"
"strings"
"testing"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
)
func TestFilesystemRepository_GetPromptDefinition(t *testing.T) {
tmpDir := t.TempDir()
if err := copyTree("testdata", tmpDir); err != nil {
t.Fatalf("failed to copy testdata: %v", err)
}
repo := NewFilesystemRepository(tmpDir)
ctx := context.Background()
t.Run("valid inline prompt", func(t *testing.T) {
p, err := repo.GetPromptDefinition(ctx, "valid-inline", "")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if p.ID != "valid-inline" {
t.Fatalf("unexpected id: %q", p.ID)
}
if p.Version != "1.0.0" {
t.Fatalf("unexpected version: %q", p.Version)
}
if p.OutputFormat != domain.FormatMarkdown {
t.Fatalf("unexpected output format: %q", p.OutputFormat)
}
if p.Validation.ValidationMode != domain.ValidationBasic {
t.Fatalf("unexpected validation mode: %q", p.Validation.ValidationMode)
}
if len(p.Templates) != 2 {
t.Fatalf("expected 2 messages, got %d", len(p.Templates))
}
if len(p.Inputs) != 1 {
t.Fatalf("expected 1 input, got %d", len(p.Inputs))
}
if p.Inputs[0].ContentType != "text/markdown" {
t.Fatalf("expected input content_type to be preserved, got %q", p.Inputs[0].ContentType)
}
})
t.Run("valid file-backed prompt", func(t *testing.T) {
p, err := repo.GetPromptDefinition(ctx, "valid-file-backed", "")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if len(p.Templates) != 2 {
t.Fatalf("expected 2 messages, got %d", len(p.Templates))
}
if !strings.Contains(p.Templates[1].Content, "{{input \"transcript\"}}") {
t.Fatalf("expected content_file template body to be loaded, got %q", p.Templates[1].Content)
}
if p.Templates[1].ContentFile == "" {
t.Fatal("expected ContentFile source metadata to be preserved")
}
if !filepath.IsAbs(p.Templates[1].ContentFile) {
t.Fatalf("expected resolved content_file path to be absolute, got %q", p.Templates[1].ContentFile)
}
})
t.Run("prompt with default_profile", func(t *testing.T) {
p, err := repo.GetPromptDefinition(ctx, "with-default-profile", "")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if p.DefaultProfile != "local-default" {
t.Fatalf("unexpected default profile: %q", p.DefaultProfile)
}
if len(p.Inputs) != 1 {
t.Fatalf("expected one input, got %d", len(p.Inputs))
}
if p.Inputs[0].ContentType != "" {
t.Fatalf("expected missing content_type to remain empty, got %q", p.Inputs[0].ContentType)
}
})
t.Run("version lookup", func(t *testing.T) {
_, err := repo.GetPromptDefinition(ctx, "valid-inline", "9.9.9")
if !errors.Is(err, ErrPromptDefinitionNotFound) {
t.Fatalf("expected ErrPromptDefinitionNotFound, got %v", err)
}
})
cases := []struct {
name string
id string
targetErr error
errSubstrs []string
}{
{name: "invalid YAML", id: "invalid_yaml", targetErr: ErrInvalidYAML},
{name: "missing id", id: "missing_id", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"id is required"}},
{name: "no messages", id: "no_messages", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"at least one message is required"}},
{name: "both content and content_file", id: "both_content_and_content_file", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"exactly one"}},
{name: "neither content nor content_file", id: "neither_content_nor_content_file", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"exactly one"}},
{name: "missing content_file", id: "missing_content_file", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"failed to read content_file"}},
{name: "duplicate input names", id: "duplicate_input_names", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"duplicate input name"}},
{name: "invalid validation mode", id: "invalid_validation_mode", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"invalid validation mode"}},
{name: "json_schema without schema_path", id: "json_schema_without_schema_path", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"schema_path"}},
{name: "unknown input field", id: "unknown_input_field", targetErr: ErrInvalidYAML, errSubstrs: []string{"field unknown_input_setting not found"}},
}
for _, tc := range cases {
t.Run(tc.name, func(t *testing.T) {
_, err := repo.GetPromptDefinition(ctx, tc.id, "")
if !errors.Is(err, tc.targetErr) {
t.Fatalf("expected %v, got %v", tc.targetErr, err)
}
for _, sub := range tc.errSubstrs {
if !strings.Contains(err.Error(), sub) {
t.Fatalf("expected error to contain %q, got %v", sub, err)
}
}
})
}
t.Run("prompt definition not found", func(t *testing.T) {
_, err := repo.GetPromptDefinition(ctx, "does-not-exist", "")
if !errors.Is(err, ErrPromptDefinitionNotFound) {
t.Fatalf("expected ErrPromptDefinitionNotFound, got %v", err)
}
})
}
func copyTree(src, dst string) error {
return filepath.WalkDir(src, func(path string, d fs.DirEntry, err error) error {
if err != nil {
return err
}
rel, err := filepath.Rel(src, path)
if err != nil {
return err
}
if rel == "." {
return nil
}
target := filepath.Join(dst, rel)
if d.IsDir() {
return os.MkdirAll(target, 0o755)
}
data, err := os.ReadFile(path)
if err != nil {
return err
}
return os.WriteFile(target, data, 0o644)
})
}

View File

@@ -1,10 +0,0 @@
id: both-content-and-content-file
version: "1.0.0"
messages:
- role: user
content: "Hi"
content_file: ./messages/user_prompt.tmpl
output:
format: text
validation_mode: none
repair_attempts: 0

View File

@@ -1,14 +0,0 @@
id: duplicate-input-names
version: "1.0.0"
inputs:
- name: transcript
required: true
- name: transcript
required: false
messages:
- role: user
content: "Hi"
output:
format: text
validation_mode: none
repair_attempts: 0

View File

@@ -1,9 +0,0 @@
id: invalid-validation-mode
version: "1.0.0"
messages:
- role: user
content: "Hi"
output:
format: text
validation_mode: nope
repair_attempts: 0

View File

@@ -1,9 +0,0 @@
id: invalid-yaml
version: "1.0.0"
messages:
- role: user
content: [broken
output:
format: text
validation_mode: none
repair_attempts: 0

View File

@@ -1,9 +0,0 @@
id: json-schema-without-schema-path
version: "1.0.0"
messages:
- role: user
content: "Return JSON"
output:
format: json
validation_mode: json_schema
repair_attempts: 0

View File

@@ -1,2 +0,0 @@
Use transcript:
{{input "transcript"}}

View File

@@ -1,9 +0,0 @@
id: missing-content-file
version: "1.0.0"
messages:
- role: user
content_file: ./messages/does_not_exist.tmpl
output:
format: text
validation_mode: none
repair_attempts: 0

View File

@@ -1,8 +0,0 @@
version: "1.0.0"
messages:
- role: user
content: "Hi"
output:
format: text
validation_mode: none
repair_attempts: 0

View File

@@ -1,8 +0,0 @@
id: neither-content-nor-content-file
version: "1.0.0"
messages:
- role: user
output:
format: text
validation_mode: none
repair_attempts: 0

View File

@@ -1,6 +0,0 @@
id: no-messages
version: "1.0.0"
output:
format: text
validation_mode: none
repair_attempts: 0

View File

@@ -1,13 +0,0 @@
id: unknown-input-field
version: "1.0.0"
inputs:
- name: transcript
required: true
unknown_input_setting: true
messages:
- role: user
content: "Hi"
output:
format: text
validation_mode: none
repair_attempts: 0

View File

@@ -1,14 +0,0 @@
id: valid-file-backed
version: "1.0.0"
inputs:
- name: transcript
required: true
messages:
- role: system
content: "Return markdown."
- role: user
content_file: ./messages/user_prompt.tmpl
output:
format: markdown
validation_mode: basic
repair_attempts: 0

View File

@@ -1,18 +0,0 @@
id: valid-inline
version: "1.0.0"
inputs:
- name: transcript
required: true
content_type: text/markdown
description: Transcript content
messages:
- role: system
content: "You are concise."
- role: user
content: |
Summarize:
{{input "transcript"}}
output:
format: markdown
validation_mode: basic
repair_attempts: 0

View File

@@ -1,13 +0,0 @@
id: with-default-profile
version: "1.0.0"
default_profile: local-default
inputs:
- name: transcript
required: true
messages:
- role: user
content: "Write output"
output:
format: text
validation_mode: none
repair_attempts: 0

View File

@@ -1,127 +0,0 @@
package usecase
import (
"context"
"path/filepath"
"testing"
"gitea.maximumdirect.net/eric/scriptorium/internal/artifact"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile"
"gitea.maximumdirect.net/eric/scriptorium/internal/prompt"
"gitea.maximumdirect.net/eric/scriptorium/internal/promptdef"
"gitea.maximumdirect.net/eric/scriptorium/internal/validate"
)
type integrationLLM struct{}
func (f *integrationLLM) Generate(ctx context.Context, req domain.GenerateRequest) (*domain.GenerateResponse, error) {
lastIntegrationRequest = req
return &domain.GenerateResponse{
Content: `{"summary":"Party discovered a captive scout beneath the tower.","events":[{"title":"Scout found in cellar","type":"discovery","notes":"Scout requested rescue from goblin raiders."}]}`,
Usage: domain.TokenUsage{
PromptTokens: 42,
CompletionTokens: 36,
TotalTokens: 78,
},
}, nil
}
var lastIntegrationRequest domain.GenerateRequest
func TestRunnerIntegrationWithPromptAndProfileFixturesAndValidation(t *testing.T) {
root, err := filepath.Abs(filepath.Join("..", ".."))
if err != nil {
t.Fatalf("failed to resolve repo root: %v", err)
}
promptsDir := filepath.Join(root, "prompts")
profilesDir := filepath.Join(root, "profiles")
schemasDir := filepath.Join(root, "schemas")
fixturesDir := filepath.Join(root, "examples", "fixtures")
t.Setenv("SCRIPTORIUM_API_KEY", "test-key")
runner := NewRunner(
promptdef.NewFilesystemRepository(promptsDir),
profile.NewFilesystemRepository(profilesDir),
artifact.NewCompositeReader(),
prompt.NewGoRenderer(),
&integrationLLM{},
validate.NewStandardValidator(schemasDir),
)
res, err := runner.Run(context.Background(), domain.RunRequest{
PromptID: "generic.structured_events",
Inputs: map[string]domain.ArtifactRef{
"transcript": {
Type: domain.ArtifactRefFile,
URI: filepath.Join(fixturesDir, "transcript.md"),
},
"glossary": {
Type: domain.ArtifactRefFile,
URI: filepath.Join(fixturesDir, "glossary.yml"),
},
},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if res.PromptID != "generic.structured_events" {
t.Fatalf("unexpected prompt id: %q", res.PromptID)
}
if res.SelectedProfileID != "local-quality" {
t.Fatalf("expected selected profile local-quality from prompt default, got %q", res.SelectedProfileID)
}
if res.RunID == "" {
t.Fatal("expected run id")
}
if res.PromptHash == "" {
t.Fatal("expected prompt hash")
}
if res.PromptVersion != "1.0.0" {
t.Fatalf("unexpected prompt version: %q", res.PromptVersion)
}
if res.Validation.Status != domain.ValidationPassed {
t.Fatalf("expected passed validation, got %q", res.Validation.Status)
}
if res.Validation.Mode != domain.ValidationJSONSchema {
t.Fatalf("expected json_schema mode, got %q", res.Validation.Mode)
}
if lastIntegrationRequest.StructuredOutput == nil {
t.Fatal("expected provider-level structured output request for json_schema prompt")
}
if lastIntegrationRequest.StructuredOutput.Type != domain.StructuredOutputJSONSchema {
t.Fatalf("expected structured output type json_schema, got %q", lastIntegrationRequest.StructuredOutput.Type)
}
if lastIntegrationRequest.StructuredOutput.JSONSchema == nil || lastIntegrationRequest.StructuredOutput.JSONSchema.Schema == nil {
t.Fatalf("expected structured output json_schema payload, got %+v", lastIntegrationRequest.StructuredOutput.JSONSchema)
}
if res.Artifact.ContentType != "application/json" {
t.Fatalf("expected application/json output, got %q", res.Artifact.ContentType)
}
if len(res.RawOutput) == 0 {
t.Fatal("expected raw output to be preserved")
}
if res.PromptHash == "" {
t.Fatal("expected non-empty prompt hash")
}
if len(res.InputHashes) != 2 {
t.Fatalf("expected two input hashes, got %d", len(res.InputHashes))
}
if res.InputHashes["transcript"] == "" || res.InputHashes["glossary"] == "" {
t.Fatalf("expected both input hashes to be set, got %#v", res.InputHashes)
}
if res.Usage.TotalTokens != 78 {
t.Fatalf("expected usage from fake llm, got %+v", res.Usage)
}
if res.StartTime.IsZero() || res.EndTime.IsZero() {
t.Fatal("expected start/end timestamps")
}
if res.EndTime.Before(res.StartTime) {
t.Fatalf("expected end >= start, got start=%v end=%v", res.StartTime, res.EndTime)
}
if res.Duration < 0 {
t.Fatalf("expected non-negative duration, got %s", res.Duration)
}
}

Some files were not shown because too many files have changed in this diff Show More