59 Commits

Author SHA1 Message Date
90b76ddad3 Update copyright statement
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-07-04 22:21:52 -05:00
d5b3d1e061 Remove completed documentation roadmaps 2026-07-05 03:16:11 +00:00
41083de46a Align internal documentation with architecture 2026-07-05 03:13:50 +00:00
07ac7e54c5 Expand consumer integration documentation 2026-07-05 03:09:32 +00:00
879cb021b2 Clarify HTTP operations documentation 2026-07-05 03:06:39 +00:00
574f88bd6a Refresh primary documentation references 2026-07-05 03:02:11 +00:00
d5d7a222a4 Establish canonical documentation links 2026-07-05 02:56:57 +00:00
aabd89aea7 Add a documentation update roadmap 2026-07-04 21:34:03 -05:00
9189cbfc22 Align serve usage flags 2026-07-05 00:24:17 +00:00
872c166ed7 Clarify artifact root symlink behavior 2026-07-05 00:22:54 +00:00
6742def4d3 Redact provider error bodies 2026-07-05 00:20:26 +00:00
1b39f82117 Defer profile extra params validation 2026-07-05 00:19:09 +00:00
f7d821067f Add HTTP size limits 2026-07-05 00:17:07 +00:00
a16f66cbc7 Enforce fs source containment 2026-07-05 00:10:47 +00:00
39485d87f6 Add an implementation plan to reflect follow-up findings from the audit 2026-07-04 19:04:33 -05:00
61e5b0fe58 Document cleanup verification details 2026-07-04 23:41:38 +00:00
f3c21c7d9f Remove stale domain run metadata 2026-07-04 23:39:40 +00:00
bc5f5d3731 Share validator mode handling 2026-07-04 23:38:45 +00:00
93a76f1d36 Share YAML catalog helpers 2026-07-04 23:37:07 +00:00
5c882f26a9 Restrict HTTP file artifact inputs 2026-07-04 23:34:44 +00:00
0d45ac6e3c Audit the internal package API and add a staged improvement roadmap 2026-07-04 18:26:25 -05:00
f7ad756fc3 Document public extra params validation 2026-07-04 23:22:13 +00:00
4fe11b1b2b Clarify public profile and source docs 2026-07-04 23:21:02 +00:00
296f9b1817 Make public options opaque 2026-07-04 23:19:17 +00:00
7a8516b0c6 Redact direct API keys in request formatting 2026-07-04 23:17:44 +00:00
2df2f530b3 Validate public JSON-like inputs 2026-07-04 23:15:50 +00:00
8b25ca72e5 Stop mutating supplied HTTP clients 2026-07-04 23:11:26 +00:00
fa02791fe9 Split prompt and profile load errors 2026-07-04 23:09:23 +00:00
32767b4eb4 Audit the public package API and add a staged improvement roadmap 2026-07-04 18:05:20 -05:00
e1e5351c5d Clean up completed roadmap docs 2026-07-04 17:24:02 -05:00
d60ef66f53 Implement a library profile API and built-in profile docs 2026-07-04 17:23:34 -05:00
4669b73d38 Update OpenAI-compatible auth docs 2026-07-04 17:04:25 +00:00
6f91603168 Add public asset source options 2026-07-04 17:02:09 +00:00
3ad247039b Add request API key support 2026-07-04 16:55:01 +00:00
32e2433628 Add built-in profile repository wiring 2026-07-04 16:48:18 +00:00
712c6b92b8 Add profile repository foundations 2026-07-04 16:41:28 +00:00
89cafcefec Add a roadmap, implementation plan, and built-in profile defaults for a production-ready public library package 2026-07-04 11:38:14 -05:00
1d7fac0a47 Implement fixes to the initial library facade 2026-07-04 11:32:47 -05:00
03d4f27d2b Document public library usage 2026-07-04 14:24:06 +00:00
4ac2038331 Add public run API with injectable LLM 2026-07-04 14:21:37 +00:00
14a7e7e04c Add public prepare engine API 2026-07-04 14:16:19 +00:00
5e522bad8b Add a roadmap and implementation plan for an initial public library package 2026-07-04 09:09:29 -05:00
23872dd742 Implement runtime parameter completion fixes
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-07-04 09:00:19 -05:00
7ffbf5f6ca Update woodpecker config to prepare only linux binaries 2026-07-04 08:53:30 -05:00
d0dc30fcc9 Document runtime provider parameters 2026-07-04 13:29:38 +00:00
b38f7b4dc3 Serialize runtime extra parameters outbound 2026-07-04 13:26:53 +00:00
0512995931 Allow JSON-compatible extra params 2026-07-04 13:24:27 +00:00
049a5feadb Make request execution overrides presence-aware 2026-07-04 13:20:39 +00:00
1798e9c575 Add a roadmap and implementation plan to support reasoning_effort and extra_params in outbound requests 2026-07-04 08:13:10 -05:00
5d4bc8c2b9 Remove completed feature roadmap docs 2026-07-02 20:12:10 -05:00
63fb8fc132 Implement support for OpenRouter sticky routing via a session_id variable 2026-07-02 20:08:44 -05:00
4d4bb7a121 Document prompt cache control behavior 2026-07-02 23:10:24 +00:00
5dcb3cd4fc Expose cache usage in adapters 2026-07-02 23:07:57 +00:00
efe346893c Serialize cache-controlled chat messages 2026-07-02 23:05:56 +00:00
c95d6fcfec Preserve cache control in rendered prompts 2026-07-02 23:03:50 +00:00
0badb4364d Add prompt cache control loading 2026-07-02 23:00:43 +00:00
1f63f8afbb Add a feature roadmap and implementation plan for cache_control values 2026-07-02 17:55:56 -05:00
bc099a31ad Finish cleanup roadmap follow-through
All checks were successful
ci/woodpecker/tag/release Pipeline was successful
2026-05-26 11:05:08 -05:00
4ff55221a3 Update internal docs for stable runner error reasons and serialized model fields 2026-05-26 15:00:06 +00:00
96 changed files with 10385 additions and 2235 deletions

1
.gitignore vendored
View File

@@ -1,6 +1,5 @@
# ---> Codex # ---> Codex
.codex .codex
AGENTS.md
# ---> Go # ---> Go
# If you prefer the allow list template instead of the deny list, see community template: # If you prefer the allow list template instead of the deny list, see community template:

View File

@@ -28,10 +28,6 @@ steps:
build_binary linux amd64 "" build_binary linux amd64 ""
build_binary linux arm64 "" build_binary linux arm64 ""
build_binary darwin amd64 ""
build_binary darwin arm64 ""
build_binary windows amd64 ".exe"
build_binary windows arm64 ".exe"
- name: publish-release - name: publish-release
image: woodpeckerci/plugin-release image: woodpeckerci/plugin-release

4
AGENTS.md Normal file
View File

@@ -0,0 +1,4 @@
Please carefully review the relevant documents in `docs/policy` before making any changes to this repository.
- `development.md` defines the contributor workflow for this application.
- `architecture.md` provides the canonical high-level architecture policy for this repository, and should be reviewed before writing or changing any code.
- `documentation.md` provides the canonical documentation policy for this repository, and should be reviewed before writing or changing any documentation.

View File

@@ -1,4 +1,4 @@
Copyright (c) 2026 eric. Copyright (c) 2026 Eric Rakestraw.
Redistribution and use in source and binary forms, with or without modification, are permitted provided that the following conditions are met: Redistribution and use in source and binary forms, with or without modification, are permitted provided that the following conditions are met:

View File

@@ -1,8 +1,12 @@
# scriptorium # scriptorium
Scriptorium is a config-driven prompt execution engine. Scriptorium is a narrow prompt-execution application for rendering prompt
requests, running them against OpenAI-compatible chat-completions endpoints, and
serving the same run workflow over HTTP.
It separates prompt definitions (what to generate) from execution profiles (how to call an OpenAI-compatible model endpoint), then runs or renders a prepared request from named input artifacts. It keeps prompt definitions, execution profiles, schemas, and input artifacts as
separate files so prompts can be reviewed and reused without baking model
runtime settings into application code.
## Quickstart ## Quickstart
@@ -23,14 +27,19 @@ This command renders the prepared prompt and effective runtime settings without
- [CLI reference](docs/cli.md) - [CLI reference](docs/cli.md)
- [Configuration reference](docs/config.md) - [Configuration reference](docs/config.md)
- [HTTP API reference](docs/api.md)
- [Operations guide](docs/operations.md) - [Operations guide](docs/operations.md)
- [Troubleshooting](docs/troubleshooting.md) - [Troubleshooting](docs/troubleshooting.md)
- [HTTP API integration](docs/integrations/http-api.md) - [Consumer integration overview](docs/consumers/api.md)
- [Go library package](docs/consumers/pkg-scriptorium.md)
- [Subprocess integration](docs/integrations/subprocess.md)
- [OpenAI-compatible chat integration](docs/integrations/openai-compatible-chat.md) - [OpenAI-compatible chat integration](docs/integrations/openai-compatible-chat.md)
- [Narratio subprocess integration](docs/integrations/narratio.md)
- [Architecture policy](docs/policy/architecture.md) - [Architecture policy](docs/policy/architecture.md)
## Examples ## Examples
- `examples/config.yml`
- `examples/config.full.yml`
- `examples/render-markdown-summary.sh` - `examples/render-markdown-summary.sh`
- `examples/http-run.json` - `examples/http-run.json`
- `examples/go-library/prepare`

406
convert.go Normal file
View File

@@ -0,0 +1,406 @@
package scriptorium
import (
"reflect"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
)
func toDomainRunRequest(req RunRequest) (domain.RunRequest, error) {
execution, err := toDomainExecutionTargetOverride(req.Execution)
if err != nil {
return domain.RunRequest{}, err
}
return domain.RunRequest{
PromptID: req.PromptID,
PromptVersion: req.PromptVersion,
ProfileID: req.ProfileID,
APIKey: req.APIKey,
Inputs: toDomainArtifactRefMap(req.Inputs),
Vars: copyStringMap(req.Vars),
Execution: execution,
Validation: toDomainOutputContractPtr(req.Validation),
Metadata: copyStringMap(req.Metadata),
}, nil
}
func fromDomainPreparedRun(prepared *domain.PreparedRun) *PreparedRun {
if prepared == nil {
return nil
}
return &PreparedRun{
PromptID: prepared.PromptID,
PromptVersion: prepared.PromptVersion,
PromptHash: prepared.PromptHash,
SelectedProfileID: prepared.SelectedProfileID,
EffectiveModelParams: fromDomainExecutionTarget(prepared.EffectiveModelParams),
OutputContract: fromDomainOutputContract(prepared.OutputContract),
StructuredOutput: fromDomainStructuredOutputSpec(prepared.StructuredOutput),
InputHashes: copyStringMap(prepared.InputHashes),
SessionID: prepared.SessionID,
RenderedPromptHash: prepared.RenderedPromptHash,
Messages: fromDomainRenderedMessages(prepared.Messages),
StartTime: prepared.StartTime,
EndTime: prepared.EndTime,
DurationMS: prepared.DurationMS,
}
}
func fromDomainRunResult(result *domain.RunResult) *RunResult {
if result == nil {
return nil
}
return &RunResult{
RunID: result.RunID,
Artifact: fromDomainArtifact(result.Artifact),
RawOutput: result.RawOutput,
Validation: fromDomainValidationResult(result.Validation),
PromptID: result.PromptID,
PromptVersion: result.PromptVersion,
PromptHash: result.PromptHash,
RenderedPromptHash: result.RenderedPromptHash,
SelectedProfileID: result.SelectedProfileID,
ModelName: result.ModelName,
Endpoint: result.Endpoint,
EffectiveModelParams: fromDomainExecutionTarget(result.EffectiveModelParams),
InputHashes: copyStringMap(result.InputHashes),
Usage: fromDomainTokenUsage(result.Usage),
StartTime: result.StartTime,
EndTime: result.EndTime,
Duration: result.Duration,
}
}
func fromDomainGenerateRequest(req domain.GenerateRequest) GenerateRequest {
return GenerateRequest{
Prompt: fromDomainRenderedPrompt(req.Prompt),
Target: fromDomainExecutionTarget(req.Target),
TargetPresence: fromDomainExecutionTargetPresence(req.TargetPresence),
StructuredOutput: fromDomainStructuredOutputSpec(req.StructuredOutput),
APIKey: req.Target.APIKey,
}
}
func toDomainGenerateResponse(resp *GenerateResponse) *domain.GenerateResponse {
if resp == nil {
return nil
}
return &domain.GenerateResponse{
Content: resp.Content,
Usage: toDomainTokenUsage(resp.Usage),
}
}
func fromDomainRenderedPrompt(prompt domain.RenderedPrompt) RenderedPrompt {
return RenderedPrompt{
SessionID: prompt.SessionID,
Messages: fromDomainRenderedMessages(prompt.Messages),
}
}
func toDomainArtifactRefMap(src map[string]ArtifactRef) map[string]domain.ArtifactRef {
if src == nil {
return nil
}
out := make(map[string]domain.ArtifactRef, len(src))
for k, v := range src {
out[k] = toDomainArtifactRef(v)
}
return out
}
func toDomainArtifactRef(ref ArtifactRef) domain.ArtifactRef {
return domain.ArtifactRef{
Type: domain.ArtifactRefType(ref.Type),
URI: ref.URI,
Body: ref.Body,
}
}
func fromDomainArtifact(artifact domain.Artifact) Artifact {
return Artifact{
Name: artifact.Name,
ContentType: artifact.ContentType,
Body: copyBytes(artifact.Body),
URI: artifact.URI,
Size: artifact.Size,
Hash: artifact.Hash,
}
}
func toDomainExecutionTargetOverride(override *ExecutionTargetOverride) (*domain.ExecutionTargetOverride, error) {
if override == nil {
return nil, nil
}
extraParams, err := copyPublicJSONMap(override.ExtraParams)
if err != nil {
return nil, err
}
return &domain.ExecutionTargetOverride{
Endpoint: override.Endpoint,
Model: override.Model,
Temperature: copyFloat64Ptr(override.Temperature),
MaxTokens: copyIntPtr(override.MaxTokens),
TopP: copyFloat64Ptr(override.TopP),
TimeoutSeconds: copyIntPtr(override.TimeoutSeconds),
ServiceTier: override.ServiceTier,
ReasoningEffort: override.ReasoningEffort,
APIKeyEnv: override.APIKeyEnv,
ExtraParams: extraParams,
}, nil
}
func fromDomainExecutionTarget(target domain.ExecutionTarget) ExecutionTarget {
return ExecutionTarget{
Endpoint: target.Endpoint,
Model: target.Model,
Temperature: target.Temperature,
MaxTokens: target.MaxTokens,
TopP: target.TopP,
TimeoutSeconds: target.TimeoutSeconds,
ServiceTier: target.ServiceTier,
ReasoningEffort: target.ReasoningEffort,
APIKeyEnv: target.APIKeyEnv,
ExtraParams: copyAnyMap(target.ExtraParams),
}
}
func fromDomainExecutionTargetPresence(presence domain.ExecutionTargetPresence) ExecutionTargetPresence {
return ExecutionTargetPresence{
Temperature: presence.Temperature,
MaxTokens: presence.MaxTokens,
TopP: presence.TopP,
TimeoutSeconds: presence.TimeoutSeconds,
}
}
func toDomainOutputContractPtr(contract *OutputContract) *domain.OutputContract {
if contract == nil {
return nil
}
out := toDomainOutputContract(*contract)
return &out
}
func toDomainOutputContract(contract OutputContract) domain.OutputContract {
return domain.OutputContract{
Format: domain.OutputFormat(contract.Format),
ValidationMode: domain.ValidationMode(contract.ValidationMode),
SchemaPath: contract.SchemaPath,
RepairAttempts: contract.RepairAttempts,
}
}
func fromDomainOutputContract(contract domain.OutputContract) OutputContract {
return OutputContract{
Format: OutputFormat(contract.Format),
ValidationMode: ValidationMode(contract.ValidationMode),
SchemaPath: contract.SchemaPath,
RepairAttempts: contract.RepairAttempts,
}
}
func fromDomainValidationResult(result domain.ValidationResult) ValidationResult {
return ValidationResult{
Status: ValidationStatus(result.Status),
Mode: ValidationMode(result.Mode),
Errors: copyStringSlice(result.Errors),
SchemaPath: result.SchemaPath,
RepairAttempts: result.RepairAttempts,
IsValid: result.IsValid,
}
}
func fromDomainTokenUsage(usage domain.TokenUsage) TokenUsage {
return TokenUsage{
PromptTokens: usage.PromptTokens,
CompletionTokens: usage.CompletionTokens,
TotalTokens: usage.TotalTokens,
CachedTokens: usage.CachedTokens,
CacheWriteTokens: usage.CacheWriteTokens,
}
}
func toDomainTokenUsage(usage TokenUsage) domain.TokenUsage {
return domain.TokenUsage{
PromptTokens: usage.PromptTokens,
CompletionTokens: usage.CompletionTokens,
TotalTokens: usage.TotalTokens,
CachedTokens: usage.CachedTokens,
CacheWriteTokens: usage.CacheWriteTokens,
}
}
func fromDomainRenderedMessages(messages []domain.RenderedMessage) []RenderedMessage {
if messages == nil {
return nil
}
out := make([]RenderedMessage, len(messages))
for i, msg := range messages {
out[i] = RenderedMessage{
Role: msg.Role,
Content: msg.Content,
CacheControl: fromDomainCacheControl(msg.CacheControl),
}
}
return out
}
func fromDomainCacheControl(cacheControl *domain.CacheControl) *CacheControl {
if cacheControl == nil {
return nil
}
return &CacheControl{
Type: CacheControlType(cacheControl.Type),
TTL: cacheControl.TTL,
}
}
func fromDomainStructuredOutputSpec(spec *domain.StructuredOutputSpec) *StructuredOutputSpec {
if spec == nil {
return nil
}
out := &StructuredOutputSpec{
Type: StructuredOutputType(spec.Type),
}
if spec.JSONSchema != nil {
out.JSONSchema = &StructuredOutputJSONSpec{
Name: spec.JSONSchema.Name,
Strict: spec.JSONSchema.Strict,
Schema: copyAny(spec.JSONSchema.Schema),
}
}
return out
}
func copyStringMap(src map[string]string) map[string]string {
if src == nil {
return nil
}
out := make(map[string]string, len(src))
for k, v := range src {
out[k] = v
}
return out
}
func copyAnyMap(src map[string]any) map[string]any {
if src == nil {
return nil
}
out := make(map[string]any, len(src))
for k, v := range src {
out[k] = copyAny(v)
}
return out
}
func copyAny(value any) any {
if value == nil {
return nil
}
switch v := value.(type) {
case map[string]any:
return copyAnyMap(v)
case []any:
out := make([]any, len(v))
for i, item := range v {
out[i] = copyAny(item)
}
return out
case []string:
return copyStringSlice(v)
case []byte:
return copyBytes(v)
default:
return copyReflectValue(reflect.ValueOf(value)).Interface()
}
}
func copyReflectValue(value reflect.Value) reflect.Value {
if !value.IsValid() {
return value
}
switch value.Kind() {
case reflect.Interface:
if value.IsNil() {
return reflect.Zero(value.Type())
}
copied := copyReflectValue(value.Elem())
if copied.IsValid() && copied.Type().AssignableTo(value.Type()) {
return copied
}
out := reflect.New(value.Type()).Elem()
out.Set(copied)
return out
case reflect.Pointer:
if value.IsNil() {
return reflect.Zero(value.Type())
}
out := reflect.New(value.Type().Elem())
out.Elem().Set(copyReflectValue(value.Elem()))
return out
case reflect.Map:
if value.IsNil() {
return reflect.Zero(value.Type())
}
out := reflect.MakeMapWithSize(value.Type(), value.Len())
iter := value.MapRange()
for iter.Next() {
out.SetMapIndex(copyReflectValue(iter.Key()), copyReflectValue(iter.Value()))
}
return out
case reflect.Slice:
if value.IsNil() {
return reflect.Zero(value.Type())
}
out := reflect.MakeSlice(value.Type(), value.Len(), value.Cap())
for i := 0; i < value.Len(); i++ {
out.Index(i).Set(copyReflectValue(value.Index(i)))
}
return out
case reflect.Array:
out := reflect.New(value.Type()).Elem()
for i := 0; i < value.Len(); i++ {
out.Index(i).Set(copyReflectValue(value.Index(i)))
}
return out
default:
return value
}
}
func copyStringSlice(src []string) []string {
if src == nil {
return nil
}
out := make([]string, len(src))
copy(out, src)
return out
}
func copyBytes(src []byte) []byte {
if src == nil {
return nil
}
out := make([]byte, len(src))
copy(out, src)
return out
}
func copyFloat64Ptr(src *float64) *float64 {
if src == nil {
return nil
}
v := *src
return &v
}
func copyIntPtr(src *int) *int {
if src == nil {
return nil
}
v := *src
return &v
}

297
docs/api.md Normal file
View File

@@ -0,0 +1,297 @@
# HTTP API Reference
This is the canonical public HTTP contract for Scriptorium.
Implemented route:
- `POST /v1/runs`
For CLI behavior, see [CLI reference](cli.md). For config and prompt/profile
file formats, see [Configuration reference](config.md).
The maintained request-shape example is `examples/http-run.json`. It requires a
running `serve` process with an artifact root that can read the referenced
files, plus a reachable model endpoint for full execution.
## Base URL And Deployment
`scriptorium serve` listens on `server.addr` or `serve --addr`. The default is
`:8080`.
The route path is always:
```text
/v1/runs
```
The HTTP adapter has no built-in authentication or authorization. Deploy it
behind trusted network and authentication controls.
## Media Types
- Request body: JSON object.
- Response body: JSON object.
- Response `Content-Type`: `application/json`.
Requests are decoded as JSON regardless of the request `Content-Type` header.
There are no shared query parameters.
## Request Limits
HTTP limits are configured through `server.*` config fields or `serve` flags:
- `server.max_request_bytes`: encoded JSON request body limit, including inline input bodies.
- `server.max_artifact_bytes`: file artifact limit for HTTP `file` input references.
- `server.max_response_bytes`: encoded JSON response limit, including artifact body and optional raw output.
Each limit defaults to `16777216` bytes. `0` disables that limit.
## `POST /v1/runs`
Runs one prompt request and returns the generated artifact, validation result,
and metadata.
### Request Body
```json
{
"prompt_id": "generic.markdown_summary",
"profile_id": "local-fast",
"prompt_version": "1.0.0",
"inputs": {
"transcript": {
"type": "file",
"uri": "./examples/fixtures/transcript.md"
},
"glossary": {
"type": "inline",
"body": "party:\n - Rin"
}
},
"vars": {
"session_date": "2026-05-04"
},
"model": {
"endpoint": "http://localhost:8000/v1",
"model": "gpt-4o-mini",
"temperature": 0,
"max_tokens": 800,
"top_p": 1,
"timeout_seconds": 120,
"service_tier": "priority",
"reasoning_effort": "medium",
"api_key_env": "SCRIPTORIUM_API_KEY",
"extra_params": {
"provider_option": "enabled"
}
},
"include_raw_output": false
}
```
Request fields:
| Field | Required | Description |
| --- | --- | --- |
| `prompt_id` | yes | Prompt ID. Must not be blank. |
| `prompt_version` | no | Prompt version filter. |
| `profile_id` | no | Execution profile ID. If omitted, the prompt must define `default_profile`. |
| `inputs` | yes | Object mapping prompt input names to input references. Must contain at least one entry. |
| `vars` | no | Object mapping template variable names to string values. |
| `model` | no | Runtime model override object. |
| `include_raw_output` | no | When `true`, include `raw_model_output` in the response. |
Input reference fields:
| Field | Required | Description |
| --- | --- | --- |
| `type` | yes | `file` or `inline`. |
| `uri` | for `file` | File URI/path. |
| `body` | for `inline` | Inline artifact body. |
HTTP `file` references require `server.artifact_root` or `serve
--artifact-root`. Relative file URIs resolve against that root. Absolute file
URIs are accepted only when lexically inside the root. Relative traversal and
absolute paths outside the root return `400 artifact_not_allowed`.
The containment check is lexical and does not resolve symlinks. Symlinks inside
the artifact root are followed by the operating system, including symlinks that
point outside the root. Keep the artifact root narrow and not writable by
untrusted users.
Model override fields:
| Field | Description |
| --- | --- |
| `endpoint` | Runtime endpoint override. |
| `model` | Runtime model override. |
| `temperature` | Number in range `0..2`. Explicit `0` is an override. |
| `max_tokens` | Integer greater than or equal to `0`. Explicit `0` is an override. |
| `top_p` | Number in range `0..1`. Explicit `0` is an override. |
| `timeout_seconds` | Integer greater than or equal to `0`. Explicit `0` disables the outbound client timeout. |
| `service_tier` | Provider-specific request tier. |
| `reasoning_effort` | Provider-specific reasoning setting. |
| `api_key_env` | Name of an environment variable containing the API key. |
| `extra_params` | JSON-compatible provider-specific top-level request fields. |
Raw API-key values are not accepted in HTTP payloads. A field such as
`api_key` is rejected as unknown JSON.
`extra_params` keys must not be empty and must not collide with reserved
outbound fields: `model`, `session_id`, `messages`, `temperature`,
`max_tokens`, `top_p`, `service_tier`, `reasoning_effort`, or
`response_format`.
### Strict JSON Rules
Request decoding is strict:
- malformed JSON returns `400 invalid_json`
- unknown request fields return `400 invalid_json`
- unknown `inputs` item fields return `400 invalid_json`
- unknown `model` fields return `400 invalid_json`
- trailing JSON tokens after the request object return `400 invalid_json`
- request bodies above the configured limit return `413 request_too_large`
### Success Response
Status: `200 OK`
```json
{
"artifact": {
"name": "output",
"content_type": "text/markdown",
"body": "Generated content",
"size": 17,
"hash": "..."
},
"validation": {
"status": "passed",
"mode": "basic",
"repair_attempts": 0,
"is_valid": true
},
"metadata": {
"run_id": "...",
"prompt_id": "generic.markdown_summary",
"prompt_version": "1.0.0",
"prompt_hash": "...",
"rendered_prompt_hash": "...",
"selected_profile_id": "local-fast",
"model_name": "gpt-4o-mini",
"endpoint": "http://localhost:8000/v1",
"model_params": {
"endpoint": "http://localhost:8000/v1",
"model": "gpt-4o-mini",
"temperature": 0.2,
"max_tokens": 500,
"top_p": 1,
"timeout_seconds": 90
},
"input_hashes": {
"transcript": "..."
},
"usage": {
"prompt_tokens": 11,
"completion_tokens": 22,
"total_tokens": 33,
"cached_tokens": 0,
"cache_write_tokens": 0
},
"start_time": "2026-05-04T12:00:00Z",
"end_time": "2026-05-04T12:00:01Z",
"duration_ms": 1000,
"validation_mode": "basic",
"validation_status": "passed",
"repair_attempts_used": 0
}
}
```
Response fields:
- `artifact`: generated output artifact.
- `validation`: validation result for the generated artifact.
- `metadata`: run and effective runtime metadata.
- `raw_model_output`: omitted unless `include_raw_output` is `true`.
`artifact.uri` is omitted when empty. `validation.errors` and
`validation.schema_path` are omitted when empty. `model_params.service_tier`,
`model_params.reasoning_effort`, `model_params.api_key_env`, and
`model_params.extra_params` are omitted when empty.
`metadata.usage.cached_tokens` and `metadata.usage.cache_write_tokens` are
always present as numbers. They are `0` when the provider omits compatible cache
usage fields or reports no cache activity.
### Validation Failure Response
Generated-content validation failures still return `200 OK`.
```json
{
"validation": {
"status": "failed",
"mode": "json",
"errors": ["invalid JSON: ..."],
"repair_attempts": 0,
"is_valid": false
}
}
```
The response still includes `artifact` and `metadata`.
## Error Responses
Error body shape:
```json
{
"error": {
"code": "invalid_request",
"message": "prompt_id is required"
}
}
```
Current status/code mapping:
| Status | Code | Meaning |
| --- | --- | --- |
| `400` | `invalid_json` | Malformed JSON, unknown JSON field, or trailing JSON token. |
| `400` | `invalid_request` | Missing/invalid request fields or invalid runtime overrides. |
| `400` | `profile_required` | No `profile_id` and prompt has no `default_profile`. |
| `400` | `prompt_load_failed` | Prompt definition YAML/contract failed to load. |
| `400` | `profile_load_failed` | Profile YAML/contract failed to load, including raw `api_key`. |
| `400` | `artifact_not_allowed` | HTTP file refs are disabled or requested path is outside artifact root. |
| `400` | `artifact_read_failed` | Input artifact could not be read or input ref was unsupported/invalid. |
| `400` | `prompt_render_failed` | Prompt template rendering failed. |
| `400` | `api_key_env_missing` | Selected `api_key_env` variable is unset or empty. |
| `404` | `not_found` | Route path is unknown. |
| `404` | `prompt_not_found` | Prompt ID/version was not found. |
| `404` | `profile_not_found` | Profile ID was not found. |
| `405` | `method_not_allowed` | Method is not `POST` on `/v1/runs`. |
| `413` | `request_too_large` | Encoded JSON request body exceeds configured request limit. |
| `413` | `artifact_too_large` | HTTP file input artifact exceeds configured artifact limit. |
| `413` | `response_too_large` | Encoded JSON response exceeds configured response limit. |
| `500` | `validation_runtime_failed` | Validator runtime/schema loading failed. |
| `500` | `internal_error` | Unclassified server error. |
| `502` | `llm_failed` | Outbound model request failed. |
HTTP error messages are intentionally concise and do not include sensitive
internal causes.
## Retry And Idempotency
Scriptorium does not provide idempotency keys, pagination, caching headers, or
rate limiting.
Clients may retry transport failures or `5xx` responses when their surrounding
workflow can tolerate another model call. A retry can generate different output
and incur another provider request.
## Example File
- `examples/http-run.json`

View File

@@ -10,110 +10,195 @@ go run ./cmd/scriptorium render \
--input glossary=./examples/fixtures/glossary.yml --input glossary=./examples/fixtures/glossary.yml
``` ```
`render` prepares and formats the prompt without calling an LLM. `render` prepares the prompt, loads input artifacts, resolves the execution
profile, and prints the prepared request without calling an LLM.
## Command Overview ## Command Overview
- `scriptorium run`: prepare prompt, call the configured LLM, write generated output, print a run summary. - `scriptorium run`: prepare a prompt, call the configured LLM, write generated output, and print a run summary.
- `scriptorium render`: prepare prompt only; write prepared-run output as `text` or `json`. - `scriptorium render`: prepare a prompt only; write prepared-run output as `text` or `json`.
- `scriptorium serve`: start the HTTP server. - `scriptorium serve`: start the HTTP server for `POST /v1/runs`.
Integration references: Canonical related references:
- [HTTP contract](integrations/http-api.md) - [Configuration reference](config.md)
- [Narratio subprocess contract](integrations/narratio.md) - [HTTP API reference](api.md)
- [Subprocess integration](integrations/subprocess.md)
## Common Argument Rules ## Common Rules
- `--config` is supported by `run`, `render`, and `serve`. - `--config` is supported by `run`, `render`, and `serve`.
- `run` and `render` require:
- `--prompt`
- at least one `--input`
- an effective `prompt_dir` and `profile_dir` (from flags or config)
- `serve` requires an effective `prompt_dir` and `profile_dir` (from flags or config).
- Positional arguments are rejected. - Positional arguments are rejected.
- `run` and `render` require `--prompt`, at least one `--input`, and an effective `prompt_dir`.
- `serve` requires an effective `prompt_dir`.
- `profile_dir` is optional. Without it, only built-in profiles are available.
- If `profile_dir` is set, custom profiles override built-in profiles with the same ID.
- Prompt cache control, `session_id`, structured output, and provider-specific profile fields are configured in YAML, not with CLI flags.
Config precedence is:
1. built-in defaults
2. config file values
3. CLI flags
## Flag Reference ## Flag Reference
### `scriptorium run` ### `scriptorium run`
- `--config <path>`: app config file path. ```bash
scriptorium run [flags]
```
Required through flags or config:
- `--prompt-dir <dir>`: prompt definition directory. - `--prompt-dir <dir>`: prompt definition directory.
- `--profile-dir <dir>`: profile definition directory.
Required as flags:
- `--prompt <id>`: prompt ID to execute.
- `--input name=path`: input file mapping. Repeat or use comma-separated mappings.
Optional flags:
- `--config <path>`: application config file.
- `--profile-dir <dir>`: custom profile definition directory.
- `--schema-dir <dir>`: schema base directory for `json_schema` validation. - `--schema-dir <dir>`: schema base directory for `json_schema` validation.
- `--prompt <id>`: prompt ID to execute. Required. - `--profile <id>`: execution profile override. If omitted, the prompt `default_profile` is used.
- `--prompt-id <id>`: deprecated alias for `--prompt`. - `--var name=value`: template variable mapping. Repeat or use comma-separated mappings.
- `--profile <id>`: explicit profile override. - `--out <path>`: write generated artifact body to a file instead of stdout.
- `--profile-id <id>`: deprecated alias for `--profile`.
- `--input name=path`: input mapping (repeatable, comma-separated accepted).
- `--var name=value`: template variable mapping (repeatable, comma-separated accepted).
- `--out <path>`: write artifact body to file instead of stdout.
- `--llm-base-url <url>`: runtime endpoint override. - `--llm-base-url <url>`: runtime endpoint override.
- `--model <name>`: runtime model override. - `--model <name>`: runtime model override.
- `--api-key-env <name>`: runtime API key environment-variable name override. - `--api-key-env <name>`: runtime API-key environment variable name override.
- `--temperature <float>`: runtime temperature override. - `--temperature <float>`: runtime temperature override.
- `--max-tokens <int>`: runtime max tokens override. - `--max-tokens <int>`: runtime max tokens override.
- `--top-p <float>`: runtime top-p override. - `--top-p <float>`: runtime top-p override.
- `--timeout <duration>`: runtime timeout override (Go duration syntax, for example `30s`, `2m`). - `--timeout <duration>`: runtime timeout override using Go duration syntax, such as `30s` or `2m`.
Deprecated aliases:
- `--prompt-id <id>`: alias for `--prompt`.
- `--profile-id <id>`: alias for `--profile`.
Runtime override notes:
- Omitted numeric override flags preserve the selected profile/default value.
- Explicit zero values override the selected profile/default value.
- `--timeout 0s` disables the outbound HTTP client timeout for that request.
- There is no raw API-key flag; use `--api-key-env`.
### `scriptorium render` ### `scriptorium render`
- Supports the same flags as `run`, except: ```bash
- no `--schema-dir` flag. scriptorium render [flags]
- Adds: ```
- `--format text|json`: prepared-run output format.
Required through flags or config:
- `--prompt-dir <dir>`: prompt definition directory.
Required as flags:
- `--prompt <id>`: prompt ID to render.
- `--input name=path`: input file mapping. Repeat or use comma-separated mappings.
Optional flags:
- `--config <path>`: application config file.
- `--prompt-dir <dir>`: prompt definition directory.
- `--profile-dir <dir>`: custom profile definition directory.
- `--profile <id>`: execution profile override.
- `--var name=value`: template variable mapping. Repeat or use comma-separated mappings.
- `--out <path>`: write prepared-run output to a file instead of stdout.
- `--llm-base-url <url>`: runtime endpoint override for the prepared request.
- `--model <name>`: runtime model override for the prepared request.
- `--api-key-env <name>`: runtime API-key environment variable name override.
- `--temperature <float>`: runtime temperature override.
- `--max-tokens <int>`: runtime max tokens override.
- `--top-p <float>`: runtime top-p override.
- `--timeout <duration>`: runtime timeout override using Go duration syntax.
- `--format text|json`: prepared-run output format. Defaults to config `defaults.render_format`, then `text`.
Deprecated aliases:
- `--prompt-id <id>`: alias for `--prompt`.
- `--profile-id <id>`: alias for `--profile`.
Notes: Notes:
- `render` still resolves profile and runtime settings.
- `render` still validates that `api_key_env` exists if the selected profile or overrides require it. - `render` resolves profiles, loads schemas for `json_schema` prompts, and validates `api_key_env`.
- `render` does not accept `--schema-dir`; use config `schema_dir` for render-time schema lookup.
- `render` does not call the LLM.
### `scriptorium serve` ### `scriptorium serve`
- `--config <path>`: app config file path. ```bash
scriptorium serve [flags]
```
Required through flags or config:
- `--prompt-dir <dir>`: prompt definition directory.
Optional flags:
- `--config <path>`: application config file.
- `--addr <listen-address>`: HTTP listen address. - `--addr <listen-address>`: HTTP listen address.
- `--prompt-dir <dir>`: prompt definition directory. - `--prompt-dir <dir>`: prompt definition directory.
- `--profile-dir <dir>`: profile definition directory. - `--profile-dir <dir>`: custom profile definition directory.
- `--schema-dir <dir>`: schema base directory for `json_schema` validation. - `--schema-dir <dir>`: schema base directory for `json_schema` validation.
- `--artifact-root <dir>`: base directory for HTTP `file` input references.
- `--max-request-bytes <n>`: maximum HTTP request body bytes; `0` disables the limit.
- `--max-artifact-bytes <n>`: maximum HTTP file artifact bytes; `0` disables the limit.
- `--max-response-bytes <n>`: maximum encoded HTTP response body bytes; `0` disables the limit.
Notes: Notes:
- `serve` does not accept runtime model override flags such as `--model` or `--llm-base-url`. - `serve` does not accept runtime model override flags such as `--model` or `--llm-base-url`.
- HTTP request fields and error codes are documented in the [HTTP API reference](api.md).
- HTTP `file` input references are rejected unless an artifact root is configured.
- HTTP size-limit flags affect only `serve`.
## Input And Variable Syntax ## Input And Variable Syntax
- `--input name=path` maps prompt input names to local file paths. - `--input name=path` maps prompt input names to local file paths.
- `--var name=value` maps template variable names to values. - `--var name=value` maps prompt template variables to string values.
- Both flags can be repeated. - Both flags can be repeated.
- Both flags also support comma-separated batches, for example: - Both flags also accept comma-separated mappings, such as `--input transcript=./t.md,glossary=./g.yml`.
- `--input transcript=./t.md,glossary=./g.yml` - Values may contain `=` after the first separator, such as `--var note=a=b=c`.
- `--var session_id=42,session_date=2026-05-04` - Empty names and empty values are rejected.
CLI `run` and `render` convert every `--input` mapping to a `file` artifact
reference. HTTP also supports `inline` input references; see [HTTP API
reference](api.md).
## Output Behavior ## Output Behavior
`run`: `run`:
- Writes generated artifact content to stdout by default. - Writes generated artifact content to stdout by default.
- Writes generated artifact content to `--out` when provided. - Writes generated artifact content to `--out` when provided.
- Prints run summary metadata to stderr on success. - Prints a success summary to stderr.
- Prints errors to stderr on failure. - Prints errors to stderr on failure.
`render`: `render`:
- Writes prepared-run output to stdout by default. - Writes prepared-run output to stdout by default.
- Writes prepared-run output to `--out` when provided. - Writes prepared-run output to `--out` when provided.
- Does not print a success summary line. - Does not print a success summary.
`serve`: `serve`:
- Logs startup and server errors to stderr. - Logs startup and server errors to stderr.
## Exit Codes ## Exit Codes
- `0`: success. - `0`: success.
- `1`: runtime/parse/config/load/render/generation/output-write error. - `1`: parse, config, load, render, generation, output-write, or runtime error.
- `2`: `run` completed, output was generated, but validation status is `failed`. - `2`: `run` completed and wrote output, but validation status is `failed`.
When `run` exits `2`, output may already be written to stdout or `--out`.
## Common Workflows ## Common Workflows
Render prompt inputs and template variables as JSON: Render prompt inputs and variables as JSON:
```bash ```bash
go run ./cmd/scriptorium render \ go run ./cmd/scriptorium render \
@@ -125,7 +210,7 @@ go run ./cmd/scriptorium render \
--format json --format json
``` ```
Run a prompt with profile override and file output: Run a prompt with an explicit profile and file output:
```bash ```bash
go run ./cmd/scriptorium run \ go run ./cmd/scriptorium run \
@@ -137,12 +222,12 @@ go run ./cmd/scriptorium run \
--out ./summary.md --out ./summary.md
``` ```
Start the HTTP server with explicit config: Start the HTTP server with example config:
```bash ```bash
go run ./cmd/scriptorium serve --config ./examples/config.yml go run ./cmd/scriptorium serve --config ./examples/config.yml
``` ```
Copyable example script: Copyable maintained script:
- `examples/render-markdown-summary.sh` - `examples/render-markdown-summary.sh`

View File

@@ -2,31 +2,32 @@
## Config Discovery And Precedence ## Config Discovery And Precedence
Application settings are loaded in this order: Application settings are resolved in this order:
1. Built-in defaults 1. built-in defaults
2. `config.yml` values 2. `config.yml` values
3. CLI overrides 3. CLI overrides
When `--config` is not provided, Scriptorium searches for config files in this order: When `--config` is omitted, Scriptorium searches:
1. `/usr/local/etc/scriptorium/config.yml` 1. `/usr/local/etc/scriptorium/config.yml`
2. `/etc/scriptorium/config.yml` 2. `/etc/scriptorium/config.yml`
If neither file exists, Scriptorium continues with built-in defaults. If neither file exists, Scriptorium uses built-in defaults. When
`--config <path>` is provided, that file must exist and decode successfully.
When `--config <path>` is provided, that file is required. ## Minimal Working Config
## Minimal App Config
```yaml ```yaml
prompt_dir: ./examples/prompts prompt_dir: ./examples/prompts
profile_dir: ./examples/profiles
``` ```
This is enough to use `run` and `render` when prompt/profile files are valid. This is enough for `run` and `render` when selected prompts use built-in
profiles. Set `profile_dir` when prompts or requests use custom profiles.
## Production-Oriented App Config The maintained repository example is `examples/config.yml`.
## Production-Oriented Config
```yaml ```yaml
prompt_dir: /opt/scriptorium/prompts prompt_dir: /opt/scriptorium/prompts
@@ -35,37 +36,57 @@ schema_dir: /opt/scriptorium/schemas
server: server:
addr: 127.0.0.1:8080 addr: 127.0.0.1:8080
artifact_root: /var/lib/scriptorium/artifacts
max_request_bytes: 16777216
max_artifact_bytes: 16777216
max_response_bytes: 16777216
defaults: defaults:
render_format: text render_format: text
``` ```
## App Config File (`config.yml`) The maintained full example is `examples/config.full.yml`.
## App Config Reference
Top-level fields: Top-level fields:
- `prompt_dir` (optional): default prompt definition directory. | Field | Default | Description |
- `profile_dir` (optional): default profile definition directory. | --- | --- | --- |
- `schema_dir` (optional): base directory for schema files used by `json_schema` validation. | `prompt_dir` | unset | Directory containing prompt definition YAML files. Required effectively by `run`, `render`, and `serve`. |
- `server.addr` (optional): default listen address for `serve`. | `profile_dir` | unset | Directory containing custom profile YAML files. Built-in profiles remain available when unset. |
- `defaults.render_format` (optional): default `render` output format (`text` or `json`). | `schema_dir` | `.` | Base directory for relative JSON Schema paths. |
| `server` | `{}` | HTTP service settings used by `serve`. |
| `defaults` | `{}` | Adapter defaults. |
Built-in defaults: `server` fields:
- `schema_dir`: `.` | Field | Default | Description |
- `server.addr`: `:8080` | --- | --- | --- |
- `defaults.render_format`: `text` | `server.addr` | `:8080` | Listen address for `serve`. |
| `server.artifact_root` | unset | Base directory for HTTP `file` input references. Without it, HTTP file refs are rejected. |
| `server.max_request_bytes` | `16777216` | Maximum encoded HTTP request body bytes. `0` disables the limit. |
| `server.max_artifact_bytes` | `16777216` | Maximum HTTP file artifact bytes. `0` disables the limit. |
| `server.max_response_bytes` | `16777216` | Maximum encoded HTTP response bytes. `0` disables the limit. |
Validation behavior: `defaults` fields:
- Config decoding is strict; unknown YAML fields are rejected. | Field | Default | Description |
- Raw API key fields are not supported in `config.yml`. | --- | --- | --- |
| `defaults.render_format` | `text` | Default `render` output format: `text` or `json`. |
Config rules:
- YAML decoding is strict; unknown fields are rejected.
- HTTP size limits must be greater than or equal to `0`.
- Empty string config values are ignored.
- Raw API key fields are not supported in app config.
## Prompt Definition Files ## Prompt Definition Files
Prompt definitions are YAML files anywhere under `prompt_dir`, including nested subdirectories. Prompt definitions are YAML files anywhere under `prompt_dir`. Nested
directories are organizational; callers select prompts by YAML `id`, not file
Subdirectories are organizational only. Callers still select prompts by the YAML `id`, not by file path. For example, `prompts/dnd/recap.yaml` may still declare `id: dnd.recap`, and callers use `--prompt dnd.recap`. path.
Example: Example:
@@ -98,53 +119,76 @@ output:
repair_attempts: 0 repair_attempts: 0
``` ```
Field reference: Prompt fields:
- `id` (required): prompt identifier. | Field | Required | Description |
- `version` (required): prompt version. | --- | --- | --- |
- `default_profile` (optional): profile ID used when request does not provide `profile_id`. | `id` | yes | Prompt identifier used by `--prompt` and HTTP `prompt_id`. |
- `description` (optional): prompt description. | `version` | yes | Prompt version. |
- `inputs` (optional list): expected named inputs. | `default_profile` | no | Profile ID used when a request does not provide a profile. |
- `messages` (required list): prompt message templates. | `description` | no | Human-readable description. |
- `output` (required object): output contract. | `session_id` | no | Go-template string rendered from request vars and forwarded as provider `session_id` when non-empty. |
| `inputs` | no | Named input declarations. |
| `messages` | yes | Chat message templates. |
| `output` | yes | Output format and validation contract. |
`inputs[]` fields: `inputs[]` fields:
- `name` (required) - `name` (required)
- `required` (optional, boolean) - `required` (optional boolean)
- `content_type` (optional metadata) - `content_type` (optional metadata)
- `description` (optional) - `description` (optional)
`messages[]` fields: `messages[]` fields:
- `role` (required) - `role` (required)
- `content` or `content_file` (exactly one is required) - exactly one of `content` or `content_file`
- `cache_control` (optional)
Message rules: Message rules:
- `content_file` resolves relative to the prompt YAML file location.
- Repeated roles are allowed. - Repeated roles are allowed.
- `content_file` is resolved relative to the prompt YAML file location. - Prompt YAML decoding is strict.
- Nested prompt files keep the same relative `content_file` behavior; `./recap.user.md` next to `dnd/recap.yaml` resolves from `dnd/`. - Duplicate input names are invalid.
- Prompt decoding is strict; unknown YAML fields are rejected. - Duplicate prompt IDs are invalid for a requested ID/version.
- Duplicate prompt IDs are invalid. If multiple files declare the requested prompt ID, Scriptorium fails instead of choosing one.
`messages[].cache_control` fields:
| Field | Required | Supported values |
| --- | --- | --- |
| `type` | yes | `ephemeral` |
| `ttl` | no | `1h` |
`session_id` behavior:
- Rendered with the same variable context as message templates.
- Trimmed and omitted when empty.
- Rejected when longer than 256 Unicode code points.
- CLI callers pass variables with `--var`; HTTP callers use `vars`.
`output` fields: `output` fields:
- `format` (required): `text`, `markdown`, or `json`. | Field | Required | Supported values |
- `validation_mode` (required): `none`, `basic`, `json`, or `json_schema`. | --- | --- | --- |
- `schema_path` (required when `validation_mode: json_schema`). | `format` | yes | `text`, `markdown`, `json` |
- `repair_attempts` (required): integer `>= 0`. | `validation_mode` | yes | `none`, `basic`, `json`, `json_schema` |
| `schema_path` | only for `json_schema` | Relative to `schema_dir` unless absolute. |
| `repair_attempts` | yes | Integer greater than or equal to `0`. |
Repair behavior boundary: Repair boundary:
- `repair_attempts` is part of the prompt contract. - `repair_attempts` is part of the prompt contract.
- CLI and HTTP currently construct the runner without a repairer, so normal `run`/`serve` execution does not perform output repair attempts. - The current CLI and HTTP wiring constructs the runner without a repairer, so normal `run` and `serve` execution does not perform repair attempts.
## Profile Definition Files ## Profile Definition Files
Execution profiles are YAML files anywhere under `profile_dir`, including nested subdirectories. Execution profiles are YAML files anywhere under `profile_dir`. Nested
directories are organizational; callers select profiles by YAML `id`, not file
path.
Subdirectories are organizational only. Callers still select profiles by the YAML `id`, not by file path. For example, `profiles/local/local-quality.yaml` may still declare `id: local-quality`, and callers use `--profile local-quality`. Scriptorium also ships built-in profiles. Custom profiles override built-ins
with the same ID.
Example: Example:
@@ -158,74 +202,123 @@ top_p: 1.0
timeout_seconds: 90 timeout_seconds: 90
api_key_env: SCRIPTORIUM_API_KEY api_key_env: SCRIPTORIUM_API_KEY
service_tier: priority service_tier: priority
reasoning_effort: medium
extra_params:
provider_route: primary
``` ```
Field reference: Profile fields:
- `id` (required) | Field | Required | Description |
- `endpoint` (required) | --- | --- | --- |
- `model` (required) | `id` | yes | Profile identifier. |
- `temperature` (optional): range `0..2` | `endpoint` | yes | OpenAI-compatible base URL including `/v1`. |
- `max_tokens` (optional): `>= 0` | `model` | yes | Provider model name. |
- `top_p` (optional): range `0..1` | `temperature` | no | Range `0..2`. |
- `timeout_seconds` (optional): `>= 0` | `max_tokens` | no | Integer greater than or equal to `0`. |
- `service_tier` (optional): provider-specific request tier such as OpenRouter `flex` or `priority` | `top_p` | no | Range `0..1`. |
- `reasoning_effort` (optional) | `timeout_seconds` | no | Integer greater than or equal to `0`. |
- `api_key_env` (optional) | `service_tier` | no | Provider-specific request tier. |
- `extra_params` (optional map of strings) | `reasoning_effort` | no | Provider-specific reasoning setting. |
| `api_key_env` | no | Environment variable name containing the API key. |
| `extra_params` | no | JSON-compatible provider-specific top-level request fields. |
Execution defaults before profile/request overrides:
| Field | Default |
| --- | --- |
| `temperature` | `0.0` |
| `max_tokens` | `0` |
| `top_p` | `1.0` |
| `timeout_seconds` | `600` |
Profile rules: Profile rules:
- Profile decoding is strict; unknown YAML fields are rejected. - Profile YAML decoding is strict.
- Duplicate custom profile IDs are invalid.
- Matching custom and built-in IDs are valid override behavior.
- Raw `api_key` is rejected; use `api_key_env`. - Raw `api_key` is rejected; use `api_key_env`.
- If `api_key_env` is set, that environment variable must be set when preparing/running. - If `api_key_env` is set, the named environment variable must be set before `run`, `render`, or HTTP execution can prepare the request.
- Duplicate profile IDs are invalid. If multiple files declare the requested profile ID, Scriptorium fails instead of choosing one. - Profile numeric fields merge by non-zero value. Request overrides are presence-aware, so explicit zero values are supported through CLI flags or HTTP model overrides.
- `extra_params` keys must not be empty and must not collide with reserved outbound fields: `model`, `session_id`, `messages`, `temperature`, `max_tokens`, `top_p`, `service_tier`, `reasoning_effort`, or `response_format`.
Current outbound request behavior: Built-in profile catalog:
- The OpenAI-compatible client currently serializes: `model`, `messages`, `temperature`, `max_tokens`, `top_p`, `service_tier`, and optional `response_format` for `json_schema` prompts. | Provider | ID | Model | API key env |
- `reasoning_effort` and `extra_params` are parsed and carried in effective settings, but are not currently serialized into outbound chat-completions requests. | --- | --- | --- | --- |
| aion-labs | `aion-2` | `aion-labs/aion-2.0` | `OPENROUTER_API_KEY` |
| anthropic | `claude-fable-latest` | `~anthropic/claude-fable-latest` | `OPENROUTER_API_KEY` |
| anthropic | `claude-haiku-latest` | `~anthropic/claude-haiku-latest` | `OPENROUTER_API_KEY` |
| anthropic | `claude-opus-latest` | `~anthropic/claude-opus-latest` | `OPENROUTER_API_KEY` |
| anthropic | `claude-sonnet-latest` | `~anthropic/claude-sonnet-latest` | `OPENROUTER_API_KEY` |
| deepseek | `deepseek-3-2` | `deepseek/deepseek-v3.2` | `OPENROUTER_API_KEY` |
| deepseek | `deepseek-4-pro` | `deepseek/deepseek-v4-pro` | `OPENROUTER_API_KEY` |
| google | `gemini-2-flash` | `google/gemini-2.5-flash` | `OPENROUTER_API_KEY` |
| google | `gemini-2-flash-lite` | `google/gemini-2.5-flash-lite` | `OPENROUTER_API_KEY` |
| google | `gemini-2-pro` | `google/gemini-2.5-pro` | `OPENROUTER_API_KEY` |
| google | `gemini-3-flash-lite` | `google/gemini-3.1-flash-lite` | `OPENROUTER_API_KEY` |
| google | `gemini-flash-latest` | `~google/gemini-flash-latest` | `OPENROUTER_API_KEY` |
| google | `gemini-pro-latest` | `~google/gemini-pro-latest` | `OPENROUTER_API_KEY` |
| google | `gemma-4-31b` | `google/gemma-4-31b-it:exacto` | `OPENROUTER_API_KEY` |
| minimax | `minimax-m2` | `minimax/minimax-m2.5` | `OPENROUTER_API_KEY` |
| minimax | `minimax-m3` | `minimax/minimax-m3` | `OPENROUTER_API_KEY` |
| mistral | `mistral-large-2512` | `mistralai/mistral-large-2512` | `OPENROUTER_API_KEY` |
| mistral | `mistral-medium-3-5` | `mistralai/mistral-medium-3-5` | `OPENROUTER_API_KEY` |
| mistral | `mistral-small-3` | `mistralai/mistral-small-3.2-24b-instruct` | `OPENROUTER_API_KEY` |
| mistral | `mistral-small-4` | `mistralai/mistral-small-2603` | `OPENROUTER_API_KEY` |
| nvidia | `nemotron-3-ultra` | `nvidia/nemotron-3-ultra-550b-a55b` | `OPENROUTER_API_KEY` |
| openai | `gpt-5-mini` | `openai/gpt-5.4-mini` | `OPENROUTER_API_KEY` |
| openai | `gpt-5-nano` | `openai/gpt-5.4-nano` | `OPENROUTER_API_KEY` |
## Schema Behavior ## Schema Behavior
Schemas are JSON files, typically in `schema_dir`. Schemas are JSON files, typically under `schema_dir`.
Rules: Rules:
- `output.validation_mode: json_schema` requires `output.schema_path`. - `output.validation_mode: json_schema` requires `output.schema_path`.
- Relative `schema_path` values resolve from `schema_dir`, including explicit nested paths such as `dnd/structured_events.schema.json`. - Relative `schema_path` values resolve from `schema_dir`.
- Absolute `schema_path` values are used directly. - Absolute `schema_path` values are used directly.
- Scriptorium does not recursively search schemas by basename; nested schemas must be referenced by their relative path. - Nested schemas must be referenced by relative path; schemas are not searched recursively by basename.
- Missing or invalid schema documents cause runtime validation errors. - Missing or invalid schema documents are runtime validation errors.
- Invalid generated JSON causes validation status `failed` (not a runtime error). - Invalid generated JSON produces validation status `failed`, not a runtime error.
Supported artifact reference types for request inputs are `file` and `inline`. ## Artifact References
Supported request input artifact reference types are:
- `file`
- `inline`
CLI `run` and `render` create `file` references from `--input name=path`.
HTTP `file` references require `server.artifact_root` or `serve
--artifact-root`. Relative file URIs resolve under that root. Absolute paths
and relative traversal outside the root are rejected by lexical checks. Symlinks
inside the root are followed by the operating system, including symlinks that
point outside the root.
HTTP `inline` references do not require an artifact root.
## Secrets Handling ## Secrets Handling
- Keep secret values in environment variables. - Keep secret values in environment variables.
- Store only environment-variable names in profile `api_key_env`. - Store only environment-variable names in `api_key_env`.
- Do not put raw API keys in config, prompts, profiles, CLI flags, or HTTP request bodies. - Do not put raw API keys in config, prompts, profiles, CLI arguments, examples, or HTTP request bodies.
## Maintained Examples ## Maintained Examples
- App config: `examples/config.yml` - Minimal app config: `examples/config.yml`
- Full app config: `examples/config.full.yml`
- Prompt examples: `examples/prompts/` - Prompt examples: `examples/prompts/`
- Profile examples: `examples/profiles/` - Custom profile examples: `examples/profiles/`
- Schema examples: `examples/schemas/` - Schema examples: `examples/schemas/`
- Input fixtures: `examples/fixtures/` - Input fixtures: `examples/fixtures/`
- Render example script: `examples/render-markdown-summary.sh` - Render script: `examples/render-markdown-summary.sh`
- HTTP request example: `examples/http-run.json` - HTTP request-shape example: `examples/http-run.json`
Example organizational layout:
```text
examples/prompts/dnd/recap.yaml
examples/profiles/local/local-quality.yaml
examples/schemas/dnd/structured_events.schema.json
```
## Integration References ## Integration References
- [Inbound HTTP contract](integrations/http-api.md) - [CLI reference](cli.md)
- [HTTP API reference](api.md)
- [Outbound OpenAI-compatible contract](integrations/openai-compatible-chat.md) - [Outbound OpenAI-compatible contract](integrations/openai-compatible-chat.md)

122
docs/consumers/api.md Normal file
View File

@@ -0,0 +1,122 @@
# Consumer Integration Overview
This guide is for applications that call Scriptorium from another codebase.
Scriptorium exposes three integration surfaces:
| Surface | Use when |
| --- | --- |
| Go package | The consumer is Go, needs typed requests/results, or wants injected LLM clients for tests. |
| CLI subprocess | The consumer wants process isolation or is not written in Go. |
| HTTP API | The consumer needs a service boundary or remote access to `POST /v1/runs`. |
Canonical references:
- Go package: [Package scriptorium](pkg-scriptorium.md)
- CLI subprocess: [Subprocess integration](../integrations/subprocess.md)
- HTTP: [HTTP API reference](../api.md)
- File formats: [Configuration reference](../config.md)
## Required Deployment Inputs
Every integration needs operators to provide:
- prompt definitions;
- profile definitions or built-in profile IDs;
- schema files when prompts use `json_schema`;
- input artifacts or inline input bodies;
- API-key environment variables or direct per-request keys where supported.
Raw API keys do not belong in config, prompt files, profile YAML, CLI
arguments, or HTTP request bodies.
## Recommended Workflow
Use the Go package when:
- the consumer is a Go application;
- the application needs `context.Context` cancellation;
- repeated calls should avoid subprocess startup;
- tests need a fake LLM client;
- direct per-request `RunRequest.APIKey` is required.
Use the CLI subprocess when:
- the consumer is not Go;
- process isolation is useful;
- stdout/stderr separation and exit codes are enough;
- the consumer already manages local files and environment variables.
Use HTTP when:
- Scriptorium should run as a service;
- multiple clients need a shared prompt/profile deployment;
- clients can reach a trusted, protected HTTP boundary.
## Minimal Go Example
```go
engine, err := scriptorium.NewEngine(scriptorium.Config{
PromptDir: "./examples/prompts",
ProfileDir: "./examples/profiles",
SchemaDir: "./examples/schemas",
})
if err != nil {
return err
}
prepared, err := engine.Prepare(ctx, scriptorium.RunRequest{
PromptID: "generic.markdown_summary",
Inputs: map[string]scriptorium.ArtifactRef{
"transcript": scriptorium.File("./examples/fixtures/transcript.md"),
"glossary": scriptorium.File("./examples/fixtures/glossary.yml"),
},
})
if err != nil {
return err
}
_ = prepared.Messages
```
Run the maintained package example:
```bash
go run ./examples/go-library/prepare
```
## Subprocess Workflow
Invoke `scriptorium render` for preflight and `scriptorium run` for generation.
Capture stdout and stderr separately. Treat exit code `2` from `run` as a
completed generation with failed validation.
See [Subprocess integration](../integrations/subprocess.md) for the stable
invocation contract.
## HTTP Workflow
Run `scriptorium serve` behind trusted controls and send JSON requests to
`POST /v1/runs`.
Do not duplicate endpoint schemas in consumers. Use the [HTTP API
reference](../api.md) as the authoritative contract.
## Consumer Responsibilities
Consumers are responsible for:
- selecting prompt/profile IDs as deployment configuration;
- supplying all required inputs and vars;
- protecting generated artifacts and rendered prompts as sensitive data;
- deciding whether to keep output when validation fails;
- implementing retries only when another model call is acceptable.
Scriptorium does not persist run state. Retrying a failed or timed-out request
can produce different output and can incur another provider request.
## Status Behavior
- Go package methods return typed results or errors that support `errors.Is`.
- CLI `run` exits `2` when generation succeeds but validation fails.
- HTTP returns `200 OK` for generated-content validation failures and exposes the failed status in the response body.
- Runtime validation failures are errors.

View File

@@ -0,0 +1,284 @@
# Package scriptorium
Import path:
```go
import "gitea.maximumdirect.net/eric/scriptorium"
```
The root package is the public Go facade for Scriptorium's prompt prepare/run
workflow. It exposes typed requests, results, source options, injected LLM
clients, and stable public errors while keeping `internal/*` packages private.
## Intended Use Cases
Use the package when a Go application needs:
- in-process prompt preparation or execution;
- typed request/result structs;
- direct `context.Context` cancellation;
- injected/fake LLM clients for tests;
- direct per-request `RunRequest.APIKey`.
Use [Subprocess integration](../integrations/subprocess.md) or the [HTTP API](../api.md)
when a process or service boundary is preferred.
## Construct An Engine
```go
engine, err := scriptorium.NewEngine(scriptorium.Config{
PromptDir: "./examples/prompts",
ProfileDir: "./examples/profiles",
SchemaDir: "./examples/schemas",
})
if err != nil {
return err
}
```
`Config` fields:
| Field | Description |
| --- | --- |
| `PromptDir` | Prompt definition directory. Required unless `WithPromptFS` or `WithPromptFile` is used. |
| `ProfileDir` | Optional custom profile directory overlaid above built-in profiles. |
| `SchemaDir` | Schema directory. Defaults to `.` when empty. |
| `Timeout` | Default timeout for the built-in OpenAI-compatible client. |
| `HTTPClient` | Optional HTTP client for the built-in OpenAI-compatible client. |
`NewEngine` accepts `nil` options and ignores them. Invalid construction wraps
`ErrInvalidConfig`.
## Source Options
Directory fields are the compatibility path. Explicit source options override
the matching directory field.
Prompt sources:
- `WithPromptFS(fsys, root)`
- `WithPromptFile(path)`
Profile sources:
- `WithProfileFS(fsys, root)`
- `WithProfileFile(path)`
- `WithProfiles(profiles...)`
Schema sources:
- `WithSchemaFS(fsys, root)`
- `WithSchemaFile(path)`
LLM source:
- `WithLLMClient(client)`
Source behavior:
- Prompt and profile YAML use the same strict rules as directory loading.
- Prompt `content_file` values resolve relative to the prompt file.
- `fs.FS` roots are containment boundaries for prompt content files and schema paths.
- File options expose the selected file by its base name.
- Profile source precedence is in-memory profiles, then explicit profile file/FS/directory source, then built-ins.
- `WithLLMClient(nil)` returns `ErrInvalidConfig`.
## In-Memory Profiles
Use `WithProfiles` when the application already has typed model settings:
```go
profile := scriptorium.OpenAICompatibleProfile(scriptorium.OpenAICompatibleProfileConfig{
ID: "app.default",
Endpoint: "https://openrouter.ai/api/v1",
Model: "mistralai/mistral-small-3.2-24b-instruct",
APIKeyRequired: true,
})
engine, err := scriptorium.NewEngine(cfg, scriptorium.WithProfiles(profile))
```
`Profile` and `OpenAICompatibleProfileConfig` include:
- `ID`
- `Endpoint`
- `Model`
- `Temperature`
- `MaxTokens`
- `TopP`
- `TimeoutSeconds`
- `ServiceTier`
- `ReasoningEffort`
- `APIKeyRequired`
- `ExtraParams`
`WithProfiles` rejects duplicate IDs in one call. In-memory profiles do not
store raw keys. When `APIKeyRequired` is true, pass the secret on each request
with `RunRequest.APIKey`.
`ExtraParams` must be JSON-compatible: strings, booleans, finite numbers,
objects with string keys, arrays/slices, and nil. Unsupported values, non-string
map keys, non-finite floats, and cycles return `ErrInvalidConfig` for profiles
or `ErrInvalidRequest` for request overrides.
## Prepare Workflow
`Prepare` resolves prompt/profile/input/schema state and renders messages
without calling an LLM.
```go
prepared, err := engine.Prepare(ctx, scriptorium.RunRequest{
PromptID: "generic.markdown_summary",
Inputs: map[string]scriptorium.ArtifactRef{
"transcript": scriptorium.File("./examples/fixtures/transcript.md"),
"glossary": scriptorium.File("./examples/fixtures/glossary.yml"),
},
})
if err != nil {
return err
}
_ = prepared.EffectiveModelParams
```
`PreparedRun` includes prompt ID/version/hash, selected profile, effective
model params, output contract, structured-output metadata, input hashes,
rendered prompt hash, rendered messages, and timing fields. It does not include
raw API-key values, model output, validation results, or internal target
presence metadata.
## Run Workflow
`Run` calls `Prepare`, invokes the configured LLM client, builds the output
artifact, and validates the output.
```go
result, err := engine.Run(ctx, scriptorium.RunRequest{
PromptID: "generic.markdown_summary",
APIKey: apiKey,
Inputs: map[string]scriptorium.ArtifactRef{
"transcript": scriptorium.File("./examples/fixtures/transcript.md"),
"glossary": scriptorium.File("./examples/fixtures/glossary.yml"),
},
})
if err != nil {
return err
}
_ = result.Artifact
```
`RunResult` includes run ID, output artifact, raw output, validation result,
prompt/profile/model metadata, effective model params, input hashes, usage, and
timing fields.
Generated-content validation failures return a successful `RunResult` with
`Validation.Status == ValidationFailed`. Runtime/schema validation errors
return an error that matches `ErrValidation`.
## Inputs
Input helpers:
- `File(path)`: file-backed artifact reference.
- `Inline(body)`: inline artifact body.
- `InlineWithURI(uri, body)`: inline artifact body with URI metadata.
Input map keys must match the prompt's expected input names.
## Injected LLM Clients
Use `WithLLMClient` for tests or custom model integrations:
```go
type fakeLLM struct{}
func (fakeLLM) Generate(ctx context.Context, req scriptorium.GenerateRequest) (*scriptorium.GenerateResponse, error) {
return &scriptorium.GenerateResponse{
Content: "generated text",
Usage: scriptorium.TokenUsage{TotalTokens: 12},
}, nil
}
engine, err := scriptorium.NewEngine(cfg, scriptorium.WithLLMClient(fakeLLM{}))
```
Injected clients receive:
- rendered prompt;
- effective execution target;
- numeric target presence metadata;
- structured-output spec when applicable;
- direct request API key when provided.
Custom clients should not log raw prompts or API keys by default.
## Overrides And API Keys
`RunRequest` fields:
| Field | Description |
| --- | --- |
| `PromptID` | Prompt ID. |
| `PromptVersion` | Optional prompt version filter. |
| `ProfileID` | Optional profile override. |
| `APIKey` | Direct per-request API key. |
| `Inputs` | Input artifact references. |
| `Vars` | Template variables. |
| `Execution` | Per-request model overrides. |
| `Validation` | Per-request output contract override. |
| `Metadata` | Request metadata reserved for callers. |
`RunRequest.Execution` uses pointer fields for numeric values so explicit zero
overrides are preserved:
```go
zero := 0
req.Execution = &scriptorium.ExecutionTargetOverride{
MaxTokens: &zero,
}
```
Direct `RunRequest.APIKey` takes precedence over profile `api_key_env` for the
default OpenAI-compatible client. It is request-scoped, uses `json:"-"`, and is
not included in `PreparedRun` or `RunResult` JSON. Normal Go string formatting
of `RunRequest` and `GenerateRequest` reports only whether a direct key is set.
Raw API keys do not belong in profile YAML, in-memory profiles, or app config.
Avoid reflection-based debug dumps of request structs because exported fields
remain visible to tools that bypass `String` and `GoString`.
## Errors
Public methods wrap context while preserving stable sentinel checks with
`errors.Is`:
- `ErrInvalidConfig`
- `ErrInvalidRequest`
- `ErrPromptNotFound`
- `ErrProfileNotFound`
- `ErrPromptLoad`
- `ErrProfileLoad`
- `ErrArtifactLoad`
- `ErrPromptRender`
- `ErrLLMGenerate`
- `ErrValidation`
Example:
```go
if errors.Is(err, scriptorium.ErrPromptNotFound) {
return err
}
```
## Examples
Run the maintained prepare-only example from the repository root:
```bash
go run ./examples/go-library/prepare
```
See also:
- [Configuration reference](../config.md)
- [Consumer integration overview](api.md)

View File

@@ -1,192 +0,0 @@
# HTTP API Integration
## Scope
This document defines the implemented inbound HTTP contract for Scriptorium.
Current scope is only:
- `POST /v1/runs`
For CLI behavior, see the [CLI reference](../cli.md).
## Endpoint
- Method: `POST`
- Path: `/v1/runs`
- Content type: JSON request/response
Route behavior:
- unknown path: `404 not_found`
- unsupported method on `/v1/runs`: `405 method_not_allowed`
Copyable request example file:
- `examples/http-run.json`
## Request Body
```json
{
"prompt_id": "generic.structured_events",
"profile_id": "local-quality",
"prompt_version": "1.0.0",
"inputs": {
"transcript": {"type": "file", "uri": "./examples/fixtures/transcript.md"},
"glossary": {"type": "inline", "body": "party:\n - Rin"}
},
"vars": {
"session_date": "2026-05-04"
},
"model": {
"endpoint": "http://localhost:8000/v1",
"model": "gpt-4o-mini",
"temperature": 0.0,
"max_tokens": 800,
"top_p": 1.0,
"timeout_seconds": 120,
"service_tier": "priority",
"reasoning_effort": "medium",
"api_key_env": "SCRIPTORIUM_API_KEY",
"extra_params": {
"route": "primary"
}
},
"include_raw_output": false
}
```
Required fields:
- `prompt_id`
- `inputs` (must contain at least one named input)
Input reference types currently supported by runtime artifact loading:
- `file`
- `inline`
## Strict JSON Rules
Request decoding uses strict JSON field checks:
- unknown request fields are rejected with `400 invalid_json`
- unknown `model` fields are rejected with `400 invalid_json`
- raw API-key payload fields such as `api_key` are rejected as unknown fields
## Success Response
Status: `200 OK`
Response shape:
```json
{
"artifact": {
"name": "output",
"content_type": "application/json",
"body": "{\"summary\":\"...\"}",
"uri": "",
"size": 123,
"hash": "..."
},
"validation": {
"status": "passed",
"mode": "json_schema",
"errors": [],
"schema_path": "structured_events.schema.json",
"repair_attempts": 0,
"is_valid": true
},
"metadata": {
"run_id": "...",
"prompt_id": "generic.structured_events",
"prompt_version": "1.0.0",
"prompt_hash": "...",
"rendered_prompt_hash": "...",
"selected_profile_id": "local-quality",
"model_name": "gpt-4o-mini",
"endpoint": "http://localhost:8000/v1",
"model_params": {
"endpoint": "http://localhost:8000/v1",
"model": "gpt-4o-mini",
"temperature": 0,
"max_tokens": 800,
"top_p": 1,
"timeout_seconds": 120,
"service_tier": "priority",
"reasoning_effort": "medium",
"api_key_env": "SCRIPTORIUM_API_KEY",
"extra_params": {
"route": "primary"
}
},
"input_hashes": {
"transcript": "..."
},
"usage": {
"prompt_tokens": 11,
"completion_tokens": 22,
"total_tokens": 33
},
"start_time": "2026-05-04T12:00:00Z",
"end_time": "2026-05-04T12:00:01Z",
"duration_ms": 1000,
"validation_mode": "json_schema",
"validation_status": "passed",
"repair_attempts_used": 0
}
}
```
`raw_model_output` is omitted by default.
To include it, send:
- `"include_raw_output": true`
## Validation Failure Behavior
Validation content failures do not map to HTTP error status.
Behavior:
- status remains `200 OK`
- `validation.status` is `failed`
- validation errors are returned in `validation.errors`
## Error Responses
Error body shape:
```json
{
"error": {
"code": "invalid_request",
"message": "prompt_id is required"
}
}
```
Current error mapping (non-exhaustive):
- `400 invalid_json`: malformed JSON or unknown JSON fields
- `400 invalid_request`: missing/invalid request fields
- `400 profile_required`: no explicit `profile_id` and prompt has no `default_profile`
- `400 prompt_load_failed`: prompt definition invalid/unloadable
- `400 profile_load_failed`: profile invalid/unloadable
- `400 artifact_read_failed`: input artifact loading failed
- `400 prompt_render_failed`: template render failed
- `400 api_key_env_missing`: named API-key environment variable is missing
- `404 prompt_not_found`
- `404 profile_not_found`
- `502 llm_failed`: outbound model request failed
- `500 validation_runtime_failed`: validator runtime/schema-load failure
- `500 internal_error`
## Security And Deployment Note
The HTTP adapter has no built-in authentication or authorization.
Deploy behind trusted controls (for example authenticated gateway/reverse proxy and network boundaries).

View File

@@ -1,114 +0,0 @@
# Narratio Subprocess Integration
## Purpose
This document defines the supported subprocess contract for Narratio invoking Scriptorium through the public CLI.
This is a CLI contract, not an internal Go package integration.
## Supported Commands
Narratio should invoke:
- `scriptorium run`
- `scriptorium render`
Use `run` for generation.
Use `render` for preflight/debug output without LLM execution.
## Recommended Invocation Shapes
Run:
```bash
scriptorium run \
--prompt <prompt_id> \
--input transcript=<path> \
--out <artifact_path>
```
Render:
```bash
scriptorium render \
--prompt <prompt_id> \
--input transcript=<path> \
--format json
```
Narratio may add:
- `--config <path>`
- `--profile <profile_id>`
- repeatable `--input name=path`
- repeatable `--var name=value`
- runtime overrides when explicitly needed (`--model`, `--llm-base-url`, `--timeout`, etc.)
## Config And Directory Behavior
Narratio can rely on resolved app config or pass explicit paths.
- default config search order:
1. `/usr/local/etc/scriptorium/config.yml`
2. `/etc/scriptorium/config.yml`
- explicit `--config` requires file existence and valid syntax
- CLI flags override config values
## Profile Selection
Profile selection follows runner behavior:
1. explicit `--profile`
2. prompt `default_profile`
3. error if neither is available
Narratio should treat prompt/profile IDs as deployment configuration, not hardcoded logic.
## Input And Variable Contract
- Inputs use repeated `--input name=path`.
- Input names must match prompt definition input names.
- Variables use repeated `--var name=value` for small metadata values.
- Prefer file inputs for large content.
## Environment Contract
- Pass through required API-key environment variables referenced by `api_key_env`.
- Never pass raw API keys via CLI arguments.
- Keep subprocess environment scoped to required variables.
## Output And Error Handling
`run`:
- stdout: artifact body unless `--out` is used
- `--out`: writes artifact to file
- stderr: success summary and errors
`render`:
- stdout: prepared-run output unless `--out` is used
- stderr: errors
Narratio should capture stdout and stderr separately.
## Exit Status Contract
- `0`: success
- `1`: parse/config/load/render/generation/IO/runtime error
- `2`: run completed but validation failed
A `run` exit code `2` can still produce output (stdout or `--out`).
## Security Notes
- Treat generated artifacts and stderr logs as potentially sensitive.
- Avoid logging full rendered prompts by default in production contexts.
- Use controlled output paths and access controls for persisted artifacts.
## Canonical References
- CLI behavior: [CLI reference](../cli.md)
- Config behavior: [Configuration reference](../config.md)
- Operations and failure handling: [Operations guide](../operations.md), [Troubleshooting](../troubleshooting.md)

View File

@@ -26,15 +26,85 @@ Example:
Serialized JSON fields: Serialized JSON fields:
- `model` (required after fallback resolution) - `model` (required after fallback resolution)
- `messages` (role/content pairs from rendered prompt) - `session_id` (only when the rendered prompt includes a non-empty session ID)
- `temperature` (only when non-zero) - `messages` (rendered prompt messages)
- `max_tokens` (only when non-zero) - `temperature` (when non-zero, or when explicitly overridden to zero)
- `top_p` (only when non-zero) - `max_tokens` (when non-zero, or when explicitly overridden to zero)
- `top_p` (when non-zero, or when explicitly overridden to zero)
- `service_tier` (only when non-empty) - `service_tier` (only when non-empty)
- `reasoning_effort` (only when non-empty)
- `response_format` (only when structured output is provided) - `response_format` (only when structured output is provided)
- profile/request `extra_params` as additional provider-specific top-level fields
`service_tier` is provider-specific. OpenRouter currently documents request values such as `flex` and `priority`; Scriptorium forwards any non-empty configured value and lets the backend validate support. `service_tier` is provider-specific. OpenRouter currently documents request values such as `flex` and `priority`; Scriptorium forwards any non-empty configured value and lets the backend validate support.
`reasoning_effort` is provider-specific. Scriptorium forwards any non-empty configured value as top-level `reasoning_effort` and lets the backend validate support.
`extra_params` are flattened into the outbound JSON object. They are not wrapped in an `extra_params` object:
```json
{
"model": "gpt-4o-mini",
"messages": [
{
"role": "user",
"content": "rendered text"
}
],
"provider_route": "primary",
"provider_options": {
"retry_budget": 2
}
}
```
`extra_params` values must be JSON-compatible. Supported value shapes include strings, numbers, booleans, objects, and arrays.
Reserved `extra_params` keys are rejected before the HTTP request is made:
- `model`
- `session_id`
- `messages`
- `temperature`
- `max_tokens`
- `top_p`
- `service_tier`
- `reasoning_effort`
- `response_format`
Empty `extra_params` keys and values that cannot be encoded as JSON are also rejected before the HTTP request is made.
`session_id` is rendered from prompt YAML using request variables and serialized as a top-level JSON request field. Scriptorium does not send an `x-session-id` header. Empty rendered session IDs are omitted, and values longer than 256 characters are rejected before the HTTP request.
Messages without prompt cache control serialize with string `content`:
```json
{
"role": "system",
"content": "rendered text"
}
```
Messages with prompt cache control serialize as a single text content-block array:
```json
{
"role": "system",
"content": [
{
"type": "text",
"text": "rendered text",
"cache_control": {
"type": "ephemeral",
"ttl": "1h"
}
}
]
}
```
When cache-control `ttl` is unset in the prompt definition, `ttl` is omitted from the outbound payload.
Structured output is currently `json_schema` only, serialized as: Structured output is currently `json_schema` only, serialized as:
```json ```json
@@ -52,7 +122,12 @@ Structured output is currently `json_schema` only, serialized as:
## Authentication Header ## Authentication Header
If `Target.APIKeyEnv` is set: If `Target.APIKey` is set:
- set `Authorization: Bearer <value>`
- do not read `Target.APIKeyEnv`
If `Target.APIKey` is empty and `Target.APIKeyEnv` is set:
- resolve environment variable value at request time - resolve environment variable value at request time
- set `Authorization: Bearer <value>` - set `Authorization: Bearer <value>`
@@ -61,7 +136,7 @@ If the environment variable is unset/empty:
- request fails before HTTP call (`ErrInvalidRequest`) - request fails before HTTP call (`ErrInvalidRequest`)
If `Target.APIKeyEnv` is empty: If both `Target.APIKey` and `Target.APIKeyEnv` are empty:
- no `Authorization` header is sent - no `Authorization` header is sent
@@ -72,6 +147,7 @@ Base timeout comes from client configuration.
Per-request override: Per-request override:
- if `Target.TimeoutSeconds > 0`, use that value for request timeout - if `Target.TimeoutSeconds > 0`, use that value for request timeout
- if `Target.TimeoutSeconds == 0` and the value came from an explicit request override, disable the HTTP client timeout
- if `Target.TimeoutSeconds < 0`, request is rejected (`ErrInvalidRequest`) - if `Target.TimeoutSeconds < 0`, request is rejected (`ErrInvalidRequest`)
## Response Expectations ## Response Expectations
@@ -82,6 +158,13 @@ Expected successful response shape (subset used):
- `usage.prompt_tokens` - `usage.prompt_tokens`
- `usage.completion_tokens` - `usage.completion_tokens`
- `usage.total_tokens` - `usage.total_tokens`
- `usage.prompt_tokens_details.cached_tokens` (optional)
- `usage.cache_write_tokens` (optional)
Absent cache usage fields are treated as zero. Parsed cache usage is exposed through run results and adapter response surfaces as:
- `cached_tokens`
- `cache_write_tokens`
Malformed response conditions include: Malformed response conditions include:
@@ -94,15 +177,12 @@ Malformed responses return `ErrMalformedResponse`.
## Error Handling ## Error Handling
- network/request-construction failures: `ErrRequestFailed` - network/request-construction failures: `ErrRequestFailed`
- non-2xx HTTP status: `ErrUnexpectedStatus` (includes status code and trimmed response body snippet) - non-2xx HTTP status: `ErrUnexpectedStatus` (includes status code; provider response bodies are not included)
- malformed response shape/content: `ErrMalformedResponse` - malformed response shape/content: `ErrMalformedResponse`
## Unsupported Or Non-Serialized Fields ## Unsupported Or Non-Serialized Fields
The following fields may exist in profile/effective settings but are not currently serialized into outbound chat-completions payloads: The client does not serialize top-level `cache_control`.
- `reasoning_effort`
- `extra_params`
No built-in retries, tool-calls, or multi-request payload modes are implemented in this client. No built-in retries, tool-calls, or multi-request payload modes are implemented in this client.

View File

@@ -0,0 +1,130 @@
# Subprocess Integration
This document defines the supported subprocess contract for downstream
applications invoking Scriptorium through the public CLI.
This is a CLI contract. Go callers that want an in-process typed API should use
the [package guide](../consumers/pkg-scriptorium.md).
## Supported Commands
Downstream applications should invoke:
- `scriptorium render` for preflight/debug output without LLM execution.
- `scriptorium run` for generation.
`scriptorium serve` is an HTTP service command, not the recommended subprocess
contract for per-request execution.
## Recommended Invocation Shapes
Render:
```bash
scriptorium render \
--config <config_path> \
--prompt <prompt_id> \
--input transcript=<path> \
--format json
```
Run:
```bash
scriptorium run \
--config <config_path> \
--prompt <prompt_id> \
--input transcript=<path> \
--out <artifact_path>
```
Callers may add:
- `--profile <profile_id>`
- repeatable `--input name=path`
- repeatable `--var name=value`
- runtime overrides when explicitly needed, such as `--model`, `--llm-base-url`, `--api-key-env`, and `--timeout`
Do not pass raw API keys as command arguments.
## Config And Directory Behavior
Callers can rely on resolved app config or pass explicit paths.
Default config search order:
1. `/usr/local/etc/scriptorium/config.yml`
2. `/etc/scriptorium/config.yml`
Rules:
- Explicit `--config` requires file existence and valid syntax.
- CLI flags override config values.
- `run` and `render` require an effective `prompt_dir`.
- `profile_dir` is optional because built-in profiles are available.
## Profile Selection
Profile selection follows runner behavior:
1. explicit `--profile`
2. prompt `default_profile`
3. error if neither is available
Treat prompt and profile IDs as deployment configuration, not hardcoded business
logic.
## Input And Variable Contract
- Inputs use repeated `--input name=path`.
- Input names must match prompt definition input names.
- Variables use repeated `--var name=value`.
- Both flags also accept comma-separated mappings.
- Prefer file inputs for large content.
CLI inputs are file references. HTTP-only `inline` references are documented in
the [HTTP API reference](../api.md).
## Environment Contract
- Pass through required API-key environment variables referenced by `api_key_env`.
- Keep subprocess environments scoped to required variables.
- Use `--api-key-env` only to name an environment variable.
- Never pass raw API keys via argv.
## Stdout And Stderr
`run`:
- stdout: generated artifact body unless `--out` is used.
- stderr: success summary and errors.
`render`:
- stdout: prepared-run output unless `--out` is used.
- stderr: errors.
Capture stdout and stderr separately. Do not parse stderr as a stable data
format beyond exit status handling.
## Exit Status Contract
- `0`: success.
- `1`: parse, config, load, render, generation, IO, or runtime error.
- `2`: `run` completed and output was written, but validation failed.
A `run` exit code `2` can still produce output on stdout or at `--out`.
Consumers must decide whether to keep or discard that output.
## Security Notes
- Treat generated artifacts, rendered prompts, stdout, and stderr as potentially sensitive.
- Use controlled output paths and access controls for persisted artifacts.
- Avoid logging full rendered prompts or generated artifacts by default.
## Canonical References
- CLI behavior: [CLI reference](../cli.md)
- Config and file formats: [Configuration reference](../config.md)
- Operations: [Operations guide](../operations.md)
- Troubleshooting: [Troubleshooting](../troubleshooting.md)

View File

@@ -1,113 +1,81 @@
# Adapter And Repository Internals # Adapter Internals
## Purpose ## Purpose
This document describes implemented adapter/repository boundaries and their current behavior. Adapters translate external interfaces into domain requests and translate domain results back out. They wire dependencies, apply app config, and own IO concerns, but they do not make runner decisions.
Source-loading behavior belongs in `docs/internal/sources.md`. User-facing CLI, HTTP, and package contracts belong in `docs/cli.md`, `docs/api.md`, and `docs/consumers/pkg-scriptorium.md`.
## Adapter Map ## Adapter Map
- `internal/adapter/cli`: CLI command parsing, app wiring, stdout/stderr handling, exit codes. - `cmd/scriptorium`: process entrypoint.
- `internal/adapter/http`: HTTP request/response mapping for `POST /v1/runs`. - `internal/adapter/cli`: command parsing, config handoff, runner construction, stdout/stderr, exit codes.
- `internal/promptdef`: filesystem prompt-definition repository. - `internal/adapter/http`: `POST /v1/runs` request/response mapping and HTTP error/status mapping.
- `internal/profile`: filesystem execution-profile repository. - root package `scriptorium`: public Go facade over internal runner types and dependencies.
- `internal/artifact`: input artifact reader.
- `internal/prompt`: Go-template renderer. Supporting implementation packages used during adapter wiring:
- `internal/llm`: OpenAI-compatible LLM client implementation.
- `internal/validate`: output validator. - `internal/config`
- `internal/format`: prepared-run formatters for `render` output. - `internal/defaults`
- `internal/format`
- `internal/llm`
- `internal/prompt`
## Inputs And Outputs ## Inputs And Outputs
CLI adapter: CLI adapter:
- Input: process args, filesystem config/assets, environment. - Input: process args, optional config file, filesystem sources, environment variables.
- Output: exit code, stdout artifact/prepared output, stderr summaries/errors. - Output: process exit code, stdout artifact/prepared output, stderr summaries and errors.
HTTP adapter: HTTP adapter:
- Input: JSON request body (`runRequestDTO`). - Input: HTTP request method/path/headers/body for `POST /v1/runs`.
- Output: JSON success/error body with mapped status codes. - Output: JSON success or error body with mapped status code.
Filesystem repositories: Public Go facade:
- Input: prompt/profile YAML files under configured directories. - Input: typed `scriptorium.Config`, `Option`, and `RunRequest` values.
- Output: normalized domain definitions/profiles or typed errors. - Output: typed `PreparedRun` and `RunResult` values plus public sentinel errors.
Artifact reader:
- Input: `domain.ArtifactRef`.
- Output: loaded `domain.Artifact`.
LLM adapter:
- Input: `domain.GenerateRequest`.
- Output: `domain.GenerateResponse`.
Validator:
- Input: artifact body + output contract.
- Output: validation result or runtime validation error.
## Boundaries ## Boundaries
- Adapters convert external representations to domain requests and back. - Adapters convert external shapes to `domain.RunRequest` and back.
- Use-case decisions remain in `internal/usecase`. - Runner orchestration remains in `internal/usecase`.
- External dependency details stay scoped to adapter packages. - Prompt/profile/schema/artifact source rules remain in repository, validator, and artifact packages.
- LLM provider request serialization remains in `internal/llm`.
- Public package types are facade types; internal domain types do not leak across the package boundary.
## Config Fields Used ## Config Fields Used
Primary app settings consumed by adapters: Adapter app settings:
- `prompt_dir` - `prompt_dir`
- `profile_dir` - `profile_dir`
- `schema_dir` - `schema_dir`
- `server.addr` - `server.addr`
- `server.artifact_root`
- `server.max_request_bytes`
- `server.max_artifact_bytes`
- `server.max_response_bytes`
- `defaults.render_format` - `defaults.render_format`
Execution profile/request settings used through runner: Execution request/profile settings passed through the runner:
- `endpoint`, `model`, `temperature`, `max_tokens`, `top_p`, `timeout_seconds`, `service_tier`, `api_key_env`, `reasoning_effort`, `extra_params` - `endpoint`
- `model`
- `temperature`
- `max_tokens`
- `top_p`
- `timeout_seconds`
- `service_tier`
- `api_key_env`
- `reasoning_effort`
- `extra_params`
## External Dependencies CLI and HTTP preserve numeric override presence so omitted values and explicit zero values remain distinct.
- YAML decoding: `gopkg.in/yaml.v3` (strict known-fields mode in config/prompt/profile loaders). ## CLI Adapter
- JSON Schema validation: `github.com/santhosh-tekuri/jsonschema/v6`.
- HTTP client/server: Go standard library.
## Failure Behavior
Strict decoding and input checks:
- config/prompt/profile loaders reject unknown YAML fields.
- prompt/profile repositories scan nested subdirectories recursively.
- prompt/profile lookup uses YAML `id` values; subdirectory paths are organizational only.
- duplicate prompt/profile IDs are invalid and fail instead of using first-match behavior.
- HTTP DTO decoder rejects unknown JSON fields.
- raw API key payload fields are rejected by strict decoding in profile/http paths.
Artifact refs:
- Supported reference types: `inline`, `file`.
- Unsupported types return `ErrUnsupportedRefType`.
LLM adapter:
- endpoint appends `/chat/completions`.
- non-2xx responses map to request failure errors.
- malformed responses (including missing/empty first choice content) are errors.
Validator:
- `basic`, `json`, `json_schema` content failures return `ValidationFailed` results.
- schema load/compile/path failures are runtime errors.
- schema lookup uses explicit `schema_path` values relative to `schema_dir`; it does not recursively search by basename.
HTTP error mapping:
- maps domain/use-case errors to stable HTTP code + error code/message.
- avoids returning internal wrapped-cause details in response payload.
## CLI Adapter Semantics
Implemented commands: Implemented commands:
@@ -115,29 +83,73 @@ Implemented commands:
- `render` - `render`
- `serve` - `serve`
Behavior highlights: Behavior:
- `run` exit `2` indicates validation failed after generation. - `run` constructs a runner with direct filesystem artifact reading and calls `Runner.Run`.
- `render` does not call the LLM. - `render` constructs a runner and calls `Runner.Prepare`; it does not call the LLM.
- `serve` exposes HTTP handler only; no built-in auth. - `serve` constructs a restricted artifact reader and HTTP handler, then starts an unauthenticated HTTP server.
- `render` supports `--format text|json`; `render` does not expose `--schema-dir`. - `run` exits `2` when generation succeeds but validation fails.
- deprecated aliases `--prompt-id` and `--profile-id` are still accepted. - parse, runtime, and output-write errors exit `1`.
- deprecated `--prompt-id` and `--profile-id` aliases are accepted.
## Tests To Inspect Before Changing ## HTTP Adapter
Behavior:
- Accepts only `POST /v1/runs`.
- Decodes JSON strictly and rejects unknown fields and trailing JSON tokens.
- Rejects empty `prompt_id` and empty `inputs` before calling the runner.
- Does not accept raw API key values in the request body.
- Returns validation failures as `200` responses with failed validation details.
- Maps request-body, artifact, and encoded-response size failures to `413`.
- Maps domain and repository errors to stable error codes without returning wrapped internal cause text.
The HTTP adapter has no built-in authentication or authorization. Deployment controls must be provided outside the process.
## Public Go Facade
Behavior:
- `NewEngine` wires the same default runner components as CLI/HTTP unless options override them.
- Prompt, profile, and schema sources may come from directories, single files, or `fs.FS` roots.
- `WithProfiles` adds in-memory profiles ahead of file-backed and built-in profiles.
- `WithLLMClient` injects custom model behavior.
- `RunRequest.APIKey` is request-scoped and direct; it is used only for generation and is stripped from public results.
- internal errors are mapped to public sentinels in `errors.go`.
## Failure Behavior
Adapters should:
- keep external error payloads concise and stable.
- avoid leaking raw secret values.
- use sentinels and typed errors for mapping.
- preserve strict external input decoding.
- keep validation content failures distinct from runtime errors.
CLI writes human-readable summaries to stderr. HTTP writes JSON error envelopes. The public Go facade returns typed errors.
## State And Manifests
Adapters do not add durable run state.
- No adapter writes run manifests.
- No adapter implements checkpoint, skip, or resume behavior.
- CLI output files are caller-selected artifacts, not internal state.
## Tests To Inspect
- `internal/adapter/cli/run_test.go` - `internal/adapter/cli/run_test.go`
- `internal/adapter/http/handler_test.go` - `internal/adapter/http/handler_test.go`
- `internal/promptdef/repository_test.go` - `engine_test.go`
- `internal/profile/repository_test.go`
- `internal/artifact/reader_test.go`
- `internal/prompt/renderer_test.go`
- `internal/llm/openai_compatible_client_test.go`
- `internal/validate/standard_validator_test.go`
- `internal/format/prepared_run_test.go` - `internal/format/prepared_run_test.go`
- `internal/llm/openai_compatible_client_test.go`
## Architectural Invariants ## Architectural Invariants
- Adapter packages do not own runner decision logic. - Adapter packages stay thin and translation-focused.
- External request/response strictness is part of contract stability. - App config is resolved before dependency construction.
- Prepared-render output never includes resolved API key values. - External input strictness is part of contract stability.
- Outbound OpenAI-compatible request includes only currently serialized fields (`model`, `messages`, optional `temperature`, `max_tokens`, `top_p`, optional `response_format`). - CLI and HTTP construct runners without a repairer.
- HTTP endpoint details remain canonical in `docs/api.md`.
- Public Go package details remain canonical in `docs/consumers/pkg-scriptorium.md`.

View File

@@ -2,29 +2,32 @@
## Purpose ## Purpose
`internal/usecase.Runner` is the core use case orchestrator for prompt preparation and execution. `internal/usecase.Runner` is the core prompt-execution orchestrator. It prepares prompt requests, calls the configured LLM client for `Run`, validates generated output, and returns domain results.
It owns request validation, prompt/profile resolution, runtime-parameter merge, artifact loading, prompt rendering, structured-output setup, LLM invocation, output validation, and result metadata. Transport parsing, DTOs, CLI output, HTTP status mapping, and public package type conversion belong outside the runner.
## Inputs And Outputs ## Inputs And Outputs
Primary input type: Primary inputs:
- `domain.RunRequest` - `domain.RunRequest`
- repositories/readers/renderers/validators injected at construction
- `context.Context` for cancellation
Primary output types: Primary outputs:
- `domain.PreparedRun` from `Prepare` - `domain.PreparedRun` from `Prepare`
- `domain.RunResult` from `Run` - `domain.RunResult` from `Run`
- wrapped sentinel errors for adapter mapping
LLM boundary types: LLM boundary types:
- `domain.GenerateRequest` - `domain.GenerateRequest`
- `domain.GenerateResponse` - `domain.GenerateResponse`
## Boundaries ## Dependencies
`Runner` coordinates the following interfaces: `Runner` depends on package interfaces instead of concrete adapter types:
- `promptdef.Repository` - `promptdef.Repository`
- `profile.Repository` - `profile.Repository`
@@ -34,113 +37,110 @@ LLM boundary types:
- `validate.Validator` - `validate.Validator`
- optional `usecase.OutputRepairer` - optional `usecase.OutputRepairer`
Transport concerns (CLI flags, HTTP DTO parsing, status-code mapping) stay outside runner. The CLI, HTTP adapter, and public Go package construct these dependencies and pass them in.
## Config Fields Used ## Config Fields
`Runner` does not read app config files directly. `Runner` does not read app config files. Effective behavior is determined by injected dependencies and the `domain.RunRequest`.
It receives fully constructed repositories/readers/validators from adapters. Effective behavior depends on adapter wiring, including: Adapter wiring commonly reflects these app config fields:
- prompt/profile directories - `prompt_dir`
- schema base directory - `profile_dir`
- selected profile/runtime overrides in request - `schema_dir`
- `server.artifact_root`
- HTTP request/artifact/response size limits
## External Adapters Used Runtime model settings are resolved from the selected profile plus request overrides.
`Runner` works with adapter implementations via interfaces. Current wiring from CLI/HTTP uses:
- filesystem prompt/profile repositories
- composite artifact reader
- Go-template prompt renderer
- OpenAI-compatible LLM client
- standard validator
## State And Resume Behavior
`Runner` is stateless across requests.
- No durable run-state storage.
- No built-in resume/skip checkpoints.
- Each `Run`/`Prepare` executes from request inputs and current repositories.
## Failure Behavior
Key error classes surfaced from `Runner`:
- `ErrInvalidRequest`: invalid prompt/profile/request/runtime/API-key-env prerequisites.
- `ErrProfileLoad`: prompt or profile load failures.
- `ErrArtifactLoad`: artifact read failures.
- `ErrPromptRender`: template render failures.
- `ErrLLMGenerate`: model request failures.
- `ErrValidation`: validation runtime failures (including schema load/compile failures).
Validation content failures are not run errors:
- `Run` can succeed with `Validation.Status == failed`.
- CLI maps this to exit code `2`.
- HTTP returns `200` with failed validation details.
## Prepare Flow ## Prepare Flow
`Prepare` performs: `Prepare`:
1. validate request basics (prompt ID present). 1. requires a non-empty prompt ID.
2. load prompt definition by ID/version. 2. loads the prompt definition and computes its hash.
3. select profile ID: 3. selects the profile from request `profile_id`, then prompt `default_profile`.
- explicit request profile ID 4. loads the selected execution profile.
- prompt `default_profile` 5. merges built-in execution defaults, profile values, and request overrides.
- otherwise request error 6. applies request-scoped direct API key values for public Go callers.
4. load execution profile. 7. validates endpoint, model, and credential requirements.
5. merge effective runtime target: 8. resolves the output contract and JSON Schema document when required.
- built-in execution defaults 9. reads input artifacts.
- selected profile values 10. renders prompt messages and hashes the rendered prompt.
- request overrides 11. returns a prepared run without calling the LLM.
6. verify required `api_key_env` environment variable (name only; value is not returned).
7. resolve output contract and structured-output schema payload when `json_schema` mode is active.
8. read input artifacts.
9. render prompt messages.
10. compute prompt/input/render hashes and return `PreparedRun`.
`Prepare` does not call the LLM. Numeric request overrides are presence-aware: omitted values preserve the current effective value, while explicit zero values are real overrides.
## Run Flow ## Run Flow
`Run` performs: `Run`:
1. generate run ID. 1. creates a run ID and start timestamp.
2. call `Prepare`. 2. calls `Prepare`.
3. call LLM with prepared messages/effective target/structured-output spec. 3. calls the injected LLM client with rendered messages, effective target, target presence, and structured-output settings.
4. build output artifact content type from output format. 4. builds the output artifact.
5. validate output. 5. validates the output.
6. optionally attempt bounded repair when repairer is injected and contract allows it. 6. optionally attempts bounded repair when a repairer is injected and the contract permits repair.
7. return `RunResult` with artifact, raw output, validation, hashes, profile/model metadata, usage, and timestamps. 7. returns the run result with artifact, raw output, validation, hashes, selected profile/model metadata, usage, and timing.
## Repair Hook Boundary `Run` must reuse `Prepare`; prepare logic should not be duplicated elsewhere.
Repair attempts occur only when all are true: ## Validation And Repair
- repairer is injected Validation content failures are returned as successful run results with `Validation.Status == failed`. They are not runtime errors.
- `repair_attempts > 0`
Validation runtime failures, such as schema load or compile errors, return `ErrValidation`.
Repair attempts occur only when all conditions are true:
- a repairer is injected
- `repair_attempts` is greater than zero
- validation status is `failed` - validation status is `failed`
- validation mode is `json` or `json_schema` - validation mode is `json` or `json_schema`
Current production wiring boundary: CLI and HTTP wiring call `usecase.NewRunner(...)`, which does not inject a repairer. Normal CLI and HTTP execution therefore does not repair invalid output.
- CLI and HTTP adapters call `usecase.NewRunner(...)` (no repairer argument). ## Failure Behavior
- Therefore normal CLI/HTTP execution does not perform repair attempts today.
## Tests To Inspect Before Changing Stable runner sentinels include:
- `ErrInvalidRequest`
- `ErrProfileRequired`
- `ErrAPIKeyEnvMissing`
- `ErrAPIKeyRequired`
- `ErrPromptLoad`
- `ErrProfileLoad`
- `ErrArtifactLoad`
- `ErrPromptRender`
- `ErrLLMGenerate`
- `ErrValidation`
Adapters should use `errors.Is` against sentinels and lower-level repository errors instead of matching message text.
Secret values must not appear in prepared output, run results, logs, HTTP responses, or serialized public package results. The effective API-key environment-variable name may appear.
## State And Manifests
The runner is stateless across requests.
- No durable run store.
- No manifest files.
- No checkpoint, skip, or resume behavior.
- Recovery is a new request after correcting inputs, config, or environment.
## Tests To Inspect
- `internal/usecase/runner_test.go` - `internal/usecase/runner_test.go`
- `internal/usecase/integration_test.go` - `internal/usecase/integration_test.go`
- `engine_test.go`
- `internal/adapter/cli/run_test.go` - `internal/adapter/cli/run_test.go`
- `internal/adapter/http/handler_test.go` - `internal/adapter/http/handler_test.go`
## Architectural Invariants ## Architectural Invariants
- `Run` reuses `Prepare`; prepare logic is not duplicated. - Use-case decisions stay in `internal/usecase`.
- Effective API-key environment-variable name may appear; resolved secret value must not. - `Run` reuses `Prepare`.
- Structured-output schema document must load before LLM call for `json_schema` mode. - Prompt/profile/artifact/schema loading remains behind injected boundaries.
- Validation content failures are result state; validation runtime failures are errors.
- Repair loops are bounded by `repair_attempts` and repairer presence. - Repair loops are bounded by `repair_attempts` and repairer presence.
- Runner stays transport-agnostic. - Resolved secret values are never serialized or emitted.

157
docs/internal/sources.md Normal file
View File

@@ -0,0 +1,157 @@
# Source Internals
## Purpose
This document covers implemented prompt, profile, schema, artifact, and catalog source behavior. It is for developers changing loaders or source wiring.
Full user-facing YAML and config reference material belongs in `docs/config.md`.
## Prompt Definition Sources
`internal/promptdef` provides directory-backed and `fs.FS` repositories.
Behavior:
- recursively scans `.yaml` and `.yml` files.
- decodes YAML with known-fields checking.
- looks up prompts by YAML `id`, not by path.
- optionally filters by prompt `version`.
- rejects duplicate matching prompt IDs.
- requires `id`, `version`, and at least one message.
- requires each message to set exactly one of `content` or `content_file`.
- resolves filesystem `content_file` values relative to the prompt YAML file.
- resolves `fs.FS` `content_file` values inside the configured source root.
- permits prompt subdirectories only as organization; they are not part of prompt identity.
For `fs.FS` roots, absolute paths and relative traversal outside the source root are rejected by catalog path helpers.
## Profile Sources
`internal/profile` provides directory-backed, `fs.FS`, and overlay repositories. `internal/profile/builtin` embeds built-in profile YAML assets and exposes them through the same repository interface.
Behavior:
- recursively scans `.yaml` and `.yml` files.
- decodes YAML with known-fields checking.
- looks up profiles by YAML `id`, not by path.
- rejects duplicate IDs inside the same source.
- rejects raw `api_key` fields in YAML; file-backed profiles must use `api_key_env`.
- validates required `endpoint` and `model` values.
- validates numeric profile ranges.
Overlay behavior:
- custom profiles are primary.
- built-in profiles are fallback.
- fallback occurs only after a primary `ErrProfileNotFound`.
- primary validation, YAML, duplicate, and raw-key errors are returned directly.
- duplicate IDs across custom and built-in sources are allowed because the custom profile overrides the built-in one.
The public Go facade can add in-memory profiles ahead of file-backed and built-in profiles.
## Schema Sources
`internal/validate` provides:
- `StandardValidator` for filesystem paths.
- `FSValidator` for `fs.FS` roots and single-file public schema sources.
Behavior:
- `json_schema` validation requires a non-empty `schema_path`.
- filesystem schema paths resolve relative to `schema_dir` unless absolute.
- directory-backed schema lookup uses the explicit `schema_path`; it does not search recursively by basename.
- `fs.FS` schema paths must remain inside the configured source root.
- single-file schema sources match by the configured file base name.
- schema documents are loaded before the LLM call for structured output.
- JSON parse failures are validation content failures.
- schema access, decode, registration, and compile failures are runtime validation errors.
## Artifact Sources
`internal/artifact` supports two input artifact reference types:
- `inline`
- `file`
Inline behavior:
- requires a non-empty body.
- produces text/plain artifacts.
- hashes the body bytes.
Direct file behavior:
- used by CLI `run`, CLI `render`, and the public Go facade.
- requires a non-empty URI.
- reads from the process filesystem without HTTP artifact-root restrictions.
- infers content type from file extension, defaulting to text/plain.
Restricted file behavior:
- used by HTTP `serve`.
- allows inline artifacts even when no artifact root is configured.
- denies file artifacts when no artifact root is configured.
- resolves relative file URIs against `server.artifact_root`.
- accepts absolute file URIs only when they pass containment checks.
- applies `server.max_artifact_bytes` when configured.
Restricted containment is lexical. It cleans paths and checks the relative path against the configured root; it does not resolve symlinks. Symlinks inside the root are followed by the operating system, including symlinks that target files outside the root.
## Catalog Helpers
`internal/filecatalog` centralizes shared source helpers:
- recursive YAML discovery for filesystem and `fs.FS` roots.
- deterministic sorting.
- `.yaml` and `.yml` filtering.
- display paths for diagnostics.
- YAML file stems.
- `fs.FS` root cleaning and containment checks.
Repository code should use these helpers instead of reimplementing path traversal and containment rules.
## Failure Behavior
Common source failures:
- missing prompt/profile/schema/artifact files.
- invalid YAML or JSON.
- unknown YAML fields.
- duplicate prompt or profile IDs.
- prompt/profile validation errors.
- raw API key fields in profile YAML.
- unsupported artifact reference type.
- missing inline body or file URI.
- artifact outside HTTP root.
- artifact exceeding HTTP size limit.
- schema load or compile failure.
Prompt/profile repository lookup errors are mapped by adapters separately from runtime runner errors. Validation content failures remain result state; source and schema runtime failures return errors.
## State And Manifests
Source packages do not persist run state.
- No manifests are read or written.
- No source package implements skip or resume behavior.
- Source reads reflect the current filesystem or `fs.FS` state for each request.
## Tests To Inspect
- `internal/promptdef/repository_test.go`
- `internal/profile/repository_test.go`
- `internal/profile/builtin/repository_test.go`
- `internal/artifact/reader_test.go`
- `internal/validate/standard_validator_test.go`
- `internal/usecase/integration_test.go`
- `engine_test.go`
## Architectural Invariants
- Prompt/profile identity comes from YAML `id`.
- External YAML decoding remains strict.
- File-backed profile YAML never accepts raw API key values.
- Built-in profiles are fallback, not a replacement for custom source validation.
- HTTP file artifacts remain rooted by lexical containment.
- Schema runtime failures remain errors, while JSON/schema content mismatches remain validation results.

View File

@@ -2,124 +2,163 @@
## Scope ## Scope
This document covers day-to-day operation of the CLI and HTTP service for currently implemented behavior. This guide covers operating the implemented CLI commands and HTTP service. It
does not replace the [CLI reference](cli.md), [Configuration reference](config.md),
For command syntax, see [CLI reference](cli.md). For file formats and defaults, see [Configuration reference](config.md). or [HTTP API reference](api.md).
## Operational Model ## Operational Model
Scriptorium executes one request at a time per CLI invocation or HTTP request. Scriptorium executes one prompt request per CLI invocation or HTTP request.
Important boundaries: Important boundaries:
- No durable run state is stored. - No durable run state is stored.
- No built-in resume, checkpoint, archive, or backup workflow exists. - No manifest, archive, checkpoint, or built-in backup workflow is written.
- Recovery is rerun-based: fix inputs/config, then rerun. - No built-in resume behavior exists.
- Recovery is rerun-based: correct inputs, config, or environment, then run again.
## Filesystem Layout And Config ## Filesystem Layout
Scriptorium depends on: Operational deployments usually provide:
- prompt definition files (`prompt_dir`) - `prompt_dir`: prompt definition YAML files and adjacent `content_file` templates.
- execution profile files (`profile_dir`) - `profile_dir`: optional custom profile YAML files.
- optional JSON schemas (`schema_dir`) - `schema_dir`: optional JSON Schema files.
- `server.artifact_root`: optional HTTP file-input root for `serve`.
Config discovery order when `--config` is omitted: Keep these directories readable by the Scriptorium process. Keep
`server.artifact_root` narrow and not writable by untrusted users.
1. `/usr/local/etc/scriptorium/config.yml`
2. `/etc/scriptorium/config.yml`
If neither exists, built-in defaults are used. If `--config <path>` is provided, that file must exist and parse successfully.
Built-in defaults relevant to operations:
- `schema_dir: .`
- `server.addr: :8080`
- `defaults.render_format: text`
## Normal CLI Workflow ## Normal CLI Workflow
Use `render` first when you need to verify prompt resolution and runtime settings without calling a model. Use `render` before `run` when changing prompt/profile/input wiring:
Use `run` for generation. ```bash
go run ./cmd/scriptorium render \
--config ./examples/config.yml \
--prompt generic.markdown_summary \
--input transcript=./examples/fixtures/transcript.md \
--input glossary=./examples/fixtures/glossary.yml \
--format json
```
Typical sequence: Use `run` for generation after preflight:
1. Confirm prompt/profile directories resolve through config or flags. ```bash
2. Confirm required input files exist and map to prompt input names. go run ./cmd/scriptorium run \
3. Confirm required API-key environment variables are set. --config ./examples/config.yml \
4. Confirm the selected profile's model endpoint is reachable from the process environment. --prompt generic.markdown_summary \
5. Run `render` for preflight when changing prompt/profile/input wiring. --input transcript=./examples/fixtures/transcript.md \
6. Run `run` for actual generation. --input glossary=./examples/fixtures/glossary.yml \
--out ./summary.md
```
## Secrets Handling Before production runs, confirm:
Raw API keys are not accepted in config files, profile files as `api_key`, CLI flags, or HTTP request bodies. - the effective config path is the intended one;
- prompt/profile/schema directories are readable;
Operational pattern: - input file paths exist and match prompt input names;
- required API-key environment variables are set;
- Set environment variables that hold secret values. - the selected model endpoint is reachable from the process environment.
- Set profile `api_key_env` (or runtime override `api_key_env`) to the environment variable name.
- Keep process environments scoped to only required variables.
## HTTP Service Operation ## HTTP Service Operation
Start service with: Start the service with:
```bash ```bash
go run ./cmd/scriptorium serve --config ./examples/config.yml go run ./cmd/scriptorium serve --config ./examples/config.yml
``` ```
Current inbound API behavior: The implemented HTTP route is `POST /v1/runs`; request and response fields are
defined in the [HTTP API reference](api.md).
- Route: `POST /v1/runs` The maintained HTTP request-shape example is `examples/http-run.json`.
- JSON request parsing rejects unknown fields.
- Validation content failures still return `200 OK` with `validation.status: "failed"`.
Security caveat: HTTP service notes:
- Unknown JSON fields are rejected.
- `inline` input references work without an artifact root.
- `file` input references require `server.artifact_root` or `serve --artifact-root`.
- Request bodies, HTTP file input artifacts, and encoded JSON responses are size-limited.
- Validation content failures return `200 OK` with `validation.status: "failed"`.
Security boundary:
- `serve` has no built-in authentication or authorization. - `serve` has no built-in authentication or authorization.
- Deploy only behind trusted controls (private network boundary, authenticated reverse proxy, API gateway, or equivalent). - Put it behind trusted controls such as a private network, authenticated reverse proxy, or API gateway.
- Do not expose an artifact root containing unrelated sensitive files.
- Symlinks inside the artifact root are followed by the operating system.
## Secrets Handling
Raw API keys are not accepted in app config, profiles, CLI flags, or HTTP
request bodies.
Use this pattern:
1. Set an environment variable containing the secret value.
2. Store only the variable name in profile `api_key_env` or request override `api_key_env`.
3. Scope the process environment to the minimum required variables.
## Output, Logs, And Exit Codes ## Output, Logs, And Exit Codes
`run` command: `run`:
- Generated artifact body goes to stdout by default. - stdout: generated artifact body unless `--out` is used.
- `--out` writes generated artifact to a file. - stderr: summary on success, errors on failure.
- Summary metadata line is written to stderr on success. - exit `2`: generation completed and output was written, but validation failed.
- Exit code `2` means generation completed but validation failed.
`render` command: `render`:
- Prepared-run output goes to stdout by default. - stdout: prepared-run output unless `--out` is used.
- `--out` writes prepared-run output to a file. - stderr: errors.
- Exit code is `0` on success and `1` on failure. - exit `0` on success, `1` on failure.
`serve` command: `serve`:
- Startup and server errors are written to stderr. - stderr: startup and server errors.
- HTTP response body: JSON success or error envelope.
## Validation Behavior In Operations ## Validation Behavior
Validation modes (`none`, `basic`, `json`, `json_schema`) are defined by prompt output contract. Prompt `output.validation_mode` controls validation:
Operational interpretation: - `none`: skipped.
- `basic`: output body must not be empty.
- `json`: output body must parse as JSON.
- `json_schema`: output body must parse as JSON and satisfy the configured schema.
- Validation runtime errors are hard failures (`run` exit `1`; HTTP error response). Runtime/schema failures are hard failures (`run` exit `1`, HTTP error).
- Validation content failures are soft failures (`run` exit `2`; HTTP `200` with failed status). Generated-content validation failures are soft failures (`run` exit `2`, HTTP
`200 OK` with failed validation status).
A failed validation run can still produce output. Decide whether to keep or discard that output in your surrounding workflow. ## Size Limits
## Safe Recovery Steps Defaults are documented in [Configuration reference](config.md). Operationally:
For failed runs or requests: - Keep default HTTP limits unless larger payloads are measured and expected.
- Prefer `inline` HTTP inputs for small payloads.
- Prefer `file` HTTP inputs for larger local artifacts under a controlled artifact root.
- Increase `server.max_response_bytes` when generated artifacts or requested raw output are expected to be large.
- Use `0` only when another trusted layer enforces size limits.
1. Capture stderr output or HTTP error code/message. ## Maintained Examples
2. Confirm config path and directory settings.
3. Verify prompt/profile IDs and input mappings. - `examples/config.yml`
4. Verify API-key environment-variable presence when required. - `examples/config.full.yml`
5. Reproduce with `render --format json` when prompt/profile/input resolution is uncertain. - `examples/render-markdown-summary.sh`
- `examples/http-run.json`
## Safe Recovery
For failed CLI commands or HTTP requests:
1. Capture stderr or the HTTP error `code` and `message`.
2. Confirm config path and effective directory settings.
3. Verify prompt ID, profile ID, schema path, and input mappings.
4. Verify required API-key environment variables.
5. Reproduce with `render --format json` when pre-LLM resolution is uncertain.
6. Rerun after correction. 6. Rerun after correction.
Because Scriptorium does not persist run state, rerun is the canonical recovery path. Because Scriptorium does not persist run state, rerun is the supported recovery
path.

View File

@@ -11,6 +11,7 @@ Scriptorium is a narrow prompt-execution application with three entry paths:
- CLI `run` - CLI `run`
- CLI `render` - CLI `render`
- HTTP `POST /v1/runs` through `serve` - HTTP `POST /v1/runs` through `serve`
- public Go package `gitea.maximumdirect.net/eric/scriptorium`
Domain behavior is centralized in `internal/usecase` and `internal/domain`. Domain behavior is centralized in `internal/usecase` and `internal/domain`.
@@ -26,6 +27,7 @@ Domain behavior is centralized in `internal/usecase` and `internal/domain`.
Current package map: Current package map:
- root package `scriptorium`: public Go facade over engine construction, source options, request/result types, and error mapping.
- `cmd/scriptorium`: process entrypoint. - `cmd/scriptorium`: process entrypoint.
- `internal/adapter/cli`: command parsing, app wiring for CLI commands, output behavior. - `internal/adapter/cli`: command parsing, app wiring for CLI commands, output behavior.
- `internal/adapter/http`: HTTP DTO mapping and error/status mapping. - `internal/adapter/http`: HTTP DTO mapping and error/status mapping.
@@ -34,7 +36,9 @@ Current package map:
- `internal/domain`: core request/result and contract types. - `internal/domain`: core request/result and contract types.
- `internal/usecase`: `Runner` prepare/run orchestration and repair-hook boundary. - `internal/usecase`: `Runner` prepare/run orchestration and repair-hook boundary.
- `internal/promptdef`: filesystem prompt-definition repository. - `internal/promptdef`: filesystem prompt-definition repository.
- `internal/profile`: filesystem execution-profile repository. - `internal/profile`: filesystem, `fs.FS`, and overlay execution-profile repositories.
- `internal/profile/builtin`: embedded built-in execution profiles.
- `internal/filecatalog`: shared YAML discovery and `fs.FS` source helpers.
- `internal/artifact`: artifact reference readers. - `internal/artifact`: artifact reference readers.
- `internal/prompt`: template renderer. - `internal/prompt`: template renderer.
- `internal/llm`: provider-neutral LLM client interface and OpenAI-compatible implementation. - `internal/llm`: provider-neutral LLM client interface and OpenAI-compatible implementation.
@@ -45,6 +49,7 @@ Detailed component behavior is documented in:
- `docs/internal/runner.md` - `docs/internal/runner.md`
- `docs/internal/adapters.md` - `docs/internal/adapters.md`
- `docs/internal/sources.md`
## Configuration And Precedence ## Configuration And Precedence
@@ -69,9 +74,10 @@ Scriptorium has no durable run-state store.
Current external contracts: Current external contracts:
- inbound HTTP contract: `POST /v1/runs` - inbound HTTP contract: `POST /v1/runs`, documented canonically in `docs/api.md`
- outbound model contract: OpenAI-compatible chat completions subset - outbound model contract: OpenAI-compatible chat completions subset
- subprocess contract for integrators: CLI `run`/`render` - subprocess contract for integrators: CLI `run`/`render`
- public Go package contract: `docs/consumers/pkg-scriptorium.md`
Integration docs belong under `docs/integrations/`. Integration docs belong under `docs/integrations/`.

View File

@@ -4,6 +4,7 @@ This document defines contributor workflow for Scriptorium.
## Repository Layout ## Repository Layout
- root package `scriptorium`: public Go facade, options, types, and error mapping.
- `cmd/scriptorium`: application entrypoint. - `cmd/scriptorium`: application entrypoint.
- `internal/domain`: core contracts. - `internal/domain`: core contracts.
- `internal/usecase`: runner orchestration. - `internal/usecase`: runner orchestration.
@@ -13,6 +14,8 @@ This document defines contributor workflow for Scriptorium.
- `internal/defaults`: default constants. - `internal/defaults`: default constants.
- `internal/promptdef`: prompt-definition repository. - `internal/promptdef`: prompt-definition repository.
- `internal/profile`: execution-profile repository. - `internal/profile`: execution-profile repository.
- `internal/profile/builtin`: embedded built-in execution profiles.
- `internal/filecatalog`: shared source discovery and path helpers.
- `internal/artifact`: artifact readers. - `internal/artifact`: artifact readers.
- `internal/prompt`: prompt rendering. - `internal/prompt`: prompt rendering.
- `internal/llm`: LLM client interface and OpenAI-compatible implementation. - `internal/llm`: LLM client interface and OpenAI-compatible implementation.
@@ -38,7 +41,9 @@ go test ./...
Targeted test runs commonly used during changes: Targeted test runs commonly used during changes:
```bash ```bash
go test .
go test ./internal/adapter/cli ./internal/adapter/http ./internal/usecase go test ./internal/adapter/cli ./internal/adapter/http ./internal/usecase
go test ./internal/...
``` ```
## Coding Conventions ## Coding Conventions
@@ -82,7 +87,8 @@ go test ./internal/adapter/cli ./internal/adapter/http ./internal/usecase
3. Keep business decisions in `internal/usecase`. 3. Keep business decisions in `internal/usecase`.
4. Add focused adapter tests for mapping, parse, and error behavior. 4. Add focused adapter tests for mapping, parse, and error behavior.
5. Document the new/changed boundary in `docs/internal/adapters.md`. 5. Document the new/changed boundary in `docs/internal/adapters.md`.
6. If external contract changes, update `docs/integrations/` in the same change. 6. If source-loading behavior changes, update `docs/internal/sources.md`.
7. If an external contract changes, update the canonical public or integration doc in the same change.
## How To Update Prompt/Profile/Schema Assets ## How To Update Prompt/Profile/Schema Assets
@@ -99,5 +105,6 @@ When behavior changes:
2. Keep non-roadmap docs limited to implemented behavior. 2. Keep non-roadmap docs limited to implemented behavior.
3. Update links after file moves/renames. 3. Update links after file moves/renames.
4. Re-run relevant tests and smoke commands. 4. Re-run relevant tests and smoke commands.
5. For internal boundary docs, check references with `rg "docs/internal|internal/sources" docs/policy docs/internal`.
Docs work is complete only when code/tests/examples/docs agree. Docs work is complete only when code/tests/examples/docs agree.

View File

@@ -2,12 +2,13 @@
## Purpose ## Purpose
Project documentation must help four audiences: Project documentation must help five audiences:
1. users who need to run the application; 1. users who need to run the application;
2. administrators/operators who need to configure and operate it; 2. administrators/operators who need to configure and operate it;
3. developers who need to understand and change it safely; 3. developers who need to understand and change it safely;
4. LLM coding agents that need clear scope, boundaries, and invariants. 4. LLM coding agents that need clear scope, boundaries, and invariants;
5. developers and LLM coding agents integrating this project from another codebase.
Docs should be accurate, concise, task-oriented, and organized by audience. Prefer links to canonical docs over repetition. Docs should be accurate, concise, task-oriented, and organized by audience. Prefer links to canonical docs over repetition.
@@ -42,11 +43,14 @@ Canonical homes:
- project purpose and quickstart: `README.md` - project purpose and quickstart: `README.md`
- development principles: `docs/policy/architecture.md` - development principles: `docs/policy/architecture.md`
- public HTTP API reference: `docs/api.md`
- configuration reference: `docs/config.md` - configuration reference: `docs/config.md`
- CLI reference: `docs/cli.md` - CLI reference: `docs/cli.md`
- operations and recovery: `docs/operations.md` - operations and recovery: `docs/operations.md`
- troubleshooting: `docs/troubleshooting.md` - troubleshooting: `docs/troubleshooting.md`
- public API/package consumer guidance: `docs/consumers/`
- implemented internals: `docs/internal/` - implemented internals: `docs/internal/`
- external protocol, service, and file-format contracts: `docs/integrations/`
- future work: `docs/roadmap/` - future work: `docs/roadmap/`
- contributor workflow: `docs/policy/development.md` - contributor workflow: `docs/policy/development.md`
- copyable examples: `examples/` - copyable examples: `examples/`
@@ -106,7 +110,7 @@ Recommended:
- `examples/` - `examples/`
- `docs/policy/development.md` - `docs/policy/development.md`
### Modular, staged, service-oriented, or orchestration application ### Modular, service-oriented, or orchestration application
Required: Required:
- `docs/cli.md`, if CLI-based - `docs/cli.md`, if CLI-based
@@ -119,6 +123,31 @@ Recommended:
- `docs/troubleshooting.md` - `docs/troubleshooting.md`
- validated examples under `examples/` - validated examples under `examples/`
### Public HTTP API service
Required:
- `docs/api.md`
- `docs/cli.md`, if CLI-based
- `docs/config.md`, if config-driven
- `docs/operations.md`
- `docs/internal/`
- `docs/policy/development.md`
Recommended:
- `docs/troubleshooting.md`
- `docs/consumers/`, for task-oriented client integration guides
- `docs/integrations/`, for upstream/downstream service contracts
- validated examples under `examples/`
### Project with public packages or consumer APIs
Required:
- `docs/consumers/api.md`
- one `docs/consumers/pkg-<name>.md` file per public package, if public packages exist
Recommended:
- copyable consumer examples under `examples/`, if practical
## Required Documents ## Required Documents
### README.md ### README.md
@@ -159,7 +188,35 @@ It should include:
- architectural invariants; - architectural invariants;
- explicit non-goals, if useful. - explicit non-goals, if useful.
For small projects, this file may be brief. It may simply state that the project is intentionally narrow, monolithic, and dependency-light. Notably, this file should prescribe a core development *policy* that should remain unchanged as the application evolves. It is not a place for details (e.g., CLI flags) that could change over time.
The contents of `architecture.md` should be trim and concise. LLMs may be directed to review it routinely via AGENTS.md, CLAUDE.md, or similar.
### docs/api.md
**Audience:** external HTTP API consumers, developers, LLM coding agents integrating by HTTP
Required for projects whose primary public interface is HTTP.
`docs/api.md` is the canonical public HTTP API contract. It should be normative for external consumers and should not be duplicated by README, operations docs, consumer guides, or integration docs.
It should include:
1. base URL conventions;
2. authentication and authorization behavior, if implemented;
3. response envelope;
4. supported media types and content negotiation behavior;
5. shared query parameters;
6. endpoint reference grouped by route family;
7. request parameters and validation rules;
8. response fields, units, nullability, and optionality;
9. error response shape and status codes;
10. pagination, caching, rate-limit, idempotency, and retry behavior, if implemented;
11. compact request and response examples.
It must document only implemented endpoints and behavior. Planned endpoints, proposed fields, future filters, and experimental response shapes belong only under `docs/roadmap/`.
For HTTP API projects, `docs/consumers/` may provide task-oriented client integration guides, but those guides should link to `docs/api.md` for the authoritative endpoint contract.
### docs/policy/development.md ### docs/policy/development.md
@@ -175,7 +232,7 @@ It should include:
- dependency policy; - dependency policy;
- how to add config fields; - how to add config fields;
- how to add CLI flags; - how to add CLI flags;
- how to add stages/modules/adapters, if applicable; - how to add modules or adapters, if applicable;
- how to update examples; - how to update examples;
- documentation update expectations. - documentation update expectations.
@@ -216,7 +273,7 @@ Explain when commands are useful, not just their syntax.
**Audience:** administrators, operators **Audience:** administrators, operators
Required for applications that maintain state, support resume behavior, run multiple stages, write durable artifacts, use remote storage, or require recovery procedures. Required for applications that maintain state, support resume behavior, run multi-step workflows, write durable artifacts, use remote storage, or require recovery procedures.
It should cover: It should cover:
@@ -244,11 +301,40 @@ Each entry should include:
- safe fix; - safe fix;
- relevant links. - relevant links.
### docs/consumers/
**Audience:** developers and LLM coding agents integrating this project from another codebase
Required for projects with public packages, SDKs, client APIs, plugin APIs, or other application-facing integration surfaces.
This directory describes how an external codebase should consume the project's public API. It should be task-oriented and copyable where useful. It is not the place for internal implementation details or operator procedures.
For projects whose public API is HTTP, `docs/consumers/` is not required, and it should not duplicate the endpoint reference in `docs/api.md`. If present, it may provide practical integration workflows, client-specific examples, or migration notes that link back to `docs/api.md`.
`docs/consumers/api.md` should provide the consumer-facing overview and primary implementation workflow. It should include:
1. intended consumer audience and use cases;
2. required inputs supplied by operators or deployment configuration;
3. recommended public package or API workflow;
4. minimal copyable example;
5. consumer responsibilities and boundaries;
6. retry, idempotency, or status behavior, if applicable;
7. links to package-specific docs and canonical integration contracts.
Package-specific docs should be named `pkg-<name>.md` and should include:
1. import path;
2. intended use cases;
3. primary types and functions needed by consumers;
4. minimal examples;
5. validation, error, retry, and boundary behavior;
6. links to canonical file-format or wire-protocol contracts.
### docs/internal/ ### docs/internal/
**Audience:** developers, LLM coding agents **Audience:** developers, LLM coding agents
Required for modular, staged, service-oriented, or orchestration projects. Required for modular, service-oriented, or orchestration projects.
This directory describes implemented internal components. It is not the roadmap. This directory describes implemented internal components. It is not the roadmap.
@@ -289,7 +375,9 @@ Roadmap docs should not be confused with current behavior.
Required for projects that depend on external CLIs, APIs, services, protocols, or file formats where the integration contract is important to maintain. Required for projects that depend on external CLIs, APIs, services, protocols, or file formats where the integration contract is important to maintain.
This directory contains concise, versioned reference notes for external integration contracts. It should document only the parts of the external system that this project actually uses. This directory contains concise, versioned reference notes for external integration contracts. It should document only the parts of the external system that this project actually uses or exposes.
For public HTTP API services, `docs/integrations/` should document upstream, downstream, storage, protocol, or runtime contracts that the service depends on or bridges. It should not become a second copy of the public HTTP endpoint reference; that belongs in `docs/api.md`.
Use one file per integration where useful. Use one file per integration where useful.
@@ -346,8 +434,10 @@ Before merging documentation changes, verify:
- README is concise and orientation-focused. - README is concise and orientation-focused.
- `docs/policy/architecture.md` describes development principles. - `docs/policy/architecture.md` describes development principles.
- `docs/api.md` is the canonical HTTP contract for HTTP API services.
- Future work appears only under `docs/roadmap/`. - Future work appears only under `docs/roadmap/`.
- User-facing docs avoid unnecessary internals. - User-facing docs avoid unnecessary internals.
- Consumer-facing docs explain public APIs without duplicating HTTP endpoint or integration contracts.
- Developer-facing docs preserve boundaries and invariants. - Developer-facing docs preserve boundaries and invariants.
- Config examples match the schema. - Config examples match the schema.
- CLI examples match real commands and flags. - CLI examples match real commands and flags.

View File

@@ -1,519 +0,0 @@
# Code Quality And Deduplication Audit
## Executive Summary
Overall code quality is solid for the current project size. The repository follows the documented shape: thin CLI/HTTP adapters, a central `Runner`, strict config/prompt/profile loading, narrow filesystem and LLM adapters, and package tests around the important public behaviors.
The top three refactoring targets are:
1. centralize execution-target field mapping across profiles, CLI overrides, HTTP DTOs, prepared output, and LLM request serialization;
2. share filesystem repository scanning helpers between prompt and profile repositories;
3. reduce CLI command wiring duplication for app settings, runner construction, and common preflight behavior.
The codebase appears ready for a limited cleanup pass. No major architectural rewrite is warranted. The main risk before a release is maintenance drift when adding fields or policy to execution settings, repositories, or command setup.
## Repository Map Reviewed
Main areas inspected:
- `cmd/scriptorium`: process entrypoint path, via repository layout and command docs.
- `internal/adapter/cli`: `run`, `render`, `serve` parsing, config loading, app wiring, stdout/stderr behavior, summary output, exit-code policy.
- `internal/adapter/http`: request DTOs, strict JSON decoding, domain mapping, response metadata mapping, error mapping.
- `internal/config`: app config loading, defaults, strict YAML decoding, CLI override precedence.
- `internal/defaults`: built-in runtime/config/content-type defaults.
- `internal/domain`: domain request/result/profile/target/artifact/validation types.
- `internal/usecase`: `Runner.Prepare`, `Runner.Run`, execution target merge, schema structured-output setup, validation and repair policy.
- `internal/promptdef`: recursive prompt YAML loading, prompt normalization, content-file resolution, prompt output-contract validation.
- `internal/profile`: recursive profile YAML loading, strict decoding, duplicate ID handling, profile validation, raw API-key rejection.
- `internal/artifact`: `inline` and `file` artifact resolution.
- `internal/prompt`: Go template rendering and required input checks.
- `internal/llm`: OpenAI-compatible request construction, timeout/API-key handling, response parsing.
- `internal/validate`: basic/JSON/JSON Schema validation and schema document loading.
- `internal/format`: prepared-run text/JSON output.
- `examples/`: maintained example config, prompts, profiles, schemas, fixtures, HTTP/render examples.
- Package tests under `internal/**`.
Major execution paths reviewed:
- CLI `run`: parse flags/config, build `RunRequest`, construct runner, call LLM, write artifact, print summary, choose exit code.
- CLI `render`: parse flags/config, build `RunRequest`, construct runner without LLM, prepare only, format prepared run.
- CLI `serve`: parse flags/config, construct runner, expose HTTP `POST /v1/runs`.
- HTTP `POST /v1/runs`: strict decode DTO, map to `RunRequest`, map `RunResult` to JSON response.
- Runner `Prepare` and `Run`: prompt/profile/artifact/schema resolution, render, LLM generate, validation, repair hook.
Areas not deeply inspected:
- No `internal/app`, `internal/stage`, `internal/modules`, `internal/storage`, `internal/manifest`, `pkg`, or external test-suite directories exist in the current repository.
- Full test execution was not run because this is a report-only pass; inspection was static.
## High-Confidence Deduplication Opportunities
### 1. Execution Target Field Mapping Is Repeated Across Boundaries
Affected files/packages:
- `internal/domain/domain.go`
- `internal/usecase/runner.go`
- `internal/adapter/cli/run.go`
- `internal/adapter/http/dto.go`
- `internal/adapter/http/handler.go`
- `internal/llm/openai_compatible_client.go`
- `internal/format/prepared_run.go`
- tests in `internal/usecase`, `internal/adapter/http`, `internal/adapter/cli`, `internal/llm`, and `internal/format`
Duplicated or near-duplicated behavior:
- Execution settings fields are listed and copied in several places: profile-to-target conversion, target merge, CLI runtime override construction, HTTP request mapping, HTTP metadata mapping, prepared-run text output, and OpenAI-compatible request construction.
- Adding `service_tier` required coordinated edits across many of these sites, which is a strong signal that field-level mapping is too scattered.
Why it matters:
- A future profile/runtime field can easily be parsed but not sent, displayed but not merged, accepted over HTTP but not included in metadata, or tested in one command but not another.
- This is public-interface drift risk because CLI, HTTP, render output, and provider requests all expose different slices of the same effective execution target.
Recommended refactor:
- Keep transport DTOs local to adapters, but add small mapping helpers at each boundary:
- a use-case/domain helper for profile-to-target copy and execution-target merge;
- an HTTP helper such as `executionTargetFromModelOverrideDTO` and `modelParamsDTOFromExecutionTarget`;
- an LLM helper such as `openAIChatRequestFromGenerateRequest` for outbound serialization.
- Add a table-driven test that constructs an `ExecutionTarget` with every supported field and verifies the HTTP metadata DTO and LLM request payload retain the intended fields.
- Avoid reflection-based generic copying; explicit mapping is still clearer here.
Suggested tests:
- Expand runner merge tests to use an all-fields target.
- Add HTTP DTO mapping tests that fail when a domain execution field is omitted from request or metadata mapping.
- Add LLM request serialization tests for all serialized fields and explicit omission of non-serialized fields.
Risk level: high. The refactor itself is low-to-medium implementation risk, but the duplicated behavior has high drift risk.
### 2. Prompt And Profile Filesystem Repository Scanning Is Duplicated
Affected files/packages:
- `internal/promptdef/filesystem_repository.go`
- `internal/profile/filesystem_repository.go`
- repository tests in `internal/promptdef` and `internal/profile`
Duplicated or near-duplicated behavior:
- Both repositories recursively walk configured directories, filter `.yaml`/`.yml`, sort paths, compute relative paths, derive likely IDs from filenames, use strict YAML decoding, partially decode IDs to decide whether to surface malformed likely-target files, detect duplicate IDs, and include relative paths in errors.
Why it matters:
- Prompt/profile repository policy should remain aligned: nested scanning, stable ordering, duplicate ID errors, likely-target malformed file behavior, and relative-path diagnostics.
- Future changes to repository scanning or extension rules would need to be made in two packages.
Recommended refactor:
- Introduce a narrow internal helper for filesystem catalog behavior, for example `internal/filecatalog` or another small package whose scope is only:
- recursive YAML file discovery with context cancellation;
- stable sorting;
- relative path formatting;
- YAML extension checks;
- filename stem extraction.
- Keep prompt/profile-specific normalization and validation in their current packages.
- Do not make a generic repository framework.
Suggested tests:
- Add helper-level tests for recursive YAML discovery, ordering, extension filtering, and relative path output.
- Keep existing prompt/profile behavior tests unchanged to confirm public error behavior survives.
Risk level: high for drift prevention; low implementation risk if the helper stays small.
### 3. CLI Command Setup And Runner Wiring Are Repeated
Affected files/packages:
- `internal/adapter/cli/run.go`
- `internal/adapter/cli/run_test.go`
Duplicated or near-duplicated behavior:
- `run`, `render`, and `serve` each construct similar runner dependencies.
- `run` and `serve` both create an OpenAI-compatible client with default timeout.
- `run`, `render`, and `serve` all resolve app settings, clean dirs, validate required prompt/profile dirs, and create filesystem repositories/validator/renderer/readers.
- `run` and `render` share most execution request flags and request construction, with command-specific differences around `--schema-dir`, `--format`, and LLM use.
Why it matters:
- Adding or changing a dependency, default, or preflight check can drift between commands.
- The current duplication is still readable, but it is large enough that future command additions or option changes will be error-prone.
Recommended refactor:
- Add a small CLI-local wiring helper, such as `newRunnerFromDirs(promptDir, profileDir, schemaDir string, llmClient llm.Client) *usecase.Runner`.
- Add a small CLI-local resolved settings struct for common `prompt_dir`, `profile_dir`, `schema_dir`, and render-format handling.
- Preserve the existing command-specific parse functions and flag surfaces; do not hide command behavior behind a generic command framework.
Suggested tests:
- Keep parser tests for each command.
- Add one test that verifies `run`, `render`, and `serve` use the same configured prompt/profile/schema dirs by exercising config-derived dirs.
- Keep command-level success tests for `run` and `render`.
Risk level: medium. Behavior is public, but a small CLI-local helper can be behavior-preserving.
### 4. HTTP Error Mapping Depends On Error Message Substrings
Affected files/packages:
- `internal/usecase/runner.go`
- `internal/adapter/http/handler.go`
- `internal/adapter/http/handler_test.go`
Duplicated or near-duplicated behavior:
- Runner creates `ErrInvalidRequest` with human-readable details for cases such as missing profile selection and missing API-key env.
- HTTP maps some specific invalid-request cases by checking `strings.Contains(err.Error(), ...)`.
Why it matters:
- User-facing HTTP error codes can drift if a runner error message is clarified.
- This crosses package boundaries in a brittle way: HTTP should depend on stable error identity, not exact prose from `internal/usecase`.
Recommended refactor:
- Add narrow sentinel errors or typed invalid-request reasons in `internal/usecase`, for example profile-required and API-key-env-missing.
- Keep HTTP response messages stable and adapter-owned.
- Do not expose HTTP-specific error codes from the runner.
Suggested tests:
- HTTP error mapping tests should assert `profile_required` and `api_key_env_missing` via sentinel wrapping, not via message matching.
- Runner tests should assert `errors.Is` for the new reason errors.
Risk level: high for public API stability; low-to-medium implementation risk.
## Medium-Confidence Opportunities
### 1. Strict YAML Decode Setup Is Repeated
Affected files/packages:
- `internal/config/config.go`
- `internal/promptdef/filesystem_repository.go`
- `internal/profile/filesystem_repository.go`
Duplicated or near-duplicated behavior:
- Each package constructs a YAML decoder and enables `KnownFields(true)`.
Semantic differences that may be intentional:
- Prompt/profile loaders need likely-target behavior and raw `api_key` handling.
- Config loading has its own explicit/implicit file search policy.
Why it matters:
- Strict decoding is an architectural invariant. A future YAML loader could forget to enable it.
Recommended refactor:
- Consider a tiny helper for strict YAML decode from bytes.
- Keep package-specific error wrapping and partial ID decode logic local.
Suggested tests:
- Existing unknown-field tests in config, promptdef, and profile should remain.
- Add any new YAML-consuming package with an unknown-field test.
Risk level: medium.
### 2. Schema Path Resolution Is Centralized, But Schema Loading Happens Twice
Affected files/packages:
- `internal/usecase/runner.go`
- `internal/validate/standard_validator.go`
Duplicated or near-duplicated behavior:
- For `json_schema`, `Runner.Prepare` asks the validator to load the schema document for provider-level structured output.
- Later validation resolves and compiles the same schema path again.
Semantic differences that may be intentional:
- Provider structured-output payload needs the raw JSON schema document.
- Runtime validation needs a compiled schema.
Why it matters:
- The current behavior is correct, but schema file errors can surface at different phases and the same file is read more than once during `Run`.
- Future caching or schema behavior changes should have one clear owner.
Recommended refactor:
- Keep path resolution inside `internal/validate`.
- Consider a schema service/loader interface that can return both raw document and compiled schema from one path, only if schema-related work grows.
- Do not add caching unless repeated schema loads become a measured cost.
Suggested tests:
- Keep existing nested schema path tests.
- Add a test that `Prepare` fails before LLM call when structured-output schema cannot load.
Risk level: medium.
### 3. Artifact Hashing And Output Artifact Construction Are Split
Affected files/packages:
- `internal/artifact/reader.go`
- `internal/usecase/runner.go`
Duplicated or near-duplicated behavior:
- Input artifacts and generated output artifacts both compute SHA-256 hashes and fill size/body/content-type fields.
Semantic differences that may be intentional:
- Input artifacts derive content type from source extension or inline default.
- Output artifacts derive content type from prompt output format and always use the default output artifact name.
Why it matters:
- Hash algorithm and artifact metadata policy should remain consistent.
Recommended refactor:
- Consider an `artifact.Build` or `artifact.HashBody` helper only for shared hash/size construction.
- Keep source-specific content type and naming decisions local.
Suggested tests:
- Existing artifact reader tests and runner output content-type tests should cover behavior.
- Add a small helper test if a shared hash function is introduced.
Risk level: medium-low.
### 4. Domain Contains An Unsupported `s3` Artifact Reference Constant
Affected files/packages:
- `internal/domain/domain.go`
- `internal/artifact/reader.go`
- docs that state only `inline` and `file` are implemented
Duplicated or near-duplicated behavior:
- Not a duplication issue. This is a cleanup issue: `domain.ArtifactRefS3` exists, but no reader supports it and docs do not document it as implemented.
Semantic differences that may be intentional:
- The constant may be a placeholder for a future adapter.
Why it matters:
- Future-facing code outside `docs/roadmap/` can confuse maintainers and tests. It also weakens the otherwise clean "implemented behavior only" policy.
Recommended refactor:
- Remove the constant if no near-term S3 implementation is planned.
- If kept, add a code comment that it is intentionally unsupported today and ensure docs continue to state only `inline` and `file` are supported.
Suggested tests:
- Existing unsupported artifact type tests should continue to pass.
Risk level: medium-low.
### 5. Test Fixture Setup Is Repeated In CLI And Use-Case Tests
Affected files/packages:
- `internal/adapter/cli/run_test.go`
- `internal/usecase/runner_test.go`
- `internal/usecase/integration_test.go`
Duplicated or near-duplicated behavior:
- Tests repeatedly construct temp prompt/profile dirs, write prompt/profile YAML, set up fake LLMs/readers/renderers/validators, and build runners.
Semantic differences that may be intentional:
- Package-local tests avoid exporting test helpers and keep each package independent.
Why it matters:
- Repeated setup makes behavior-preserving refactors noisier and can obscure the exact behavior under test.
Recommended refactor:
- Add package-local helper builders where duplication is highest, especially in CLI command tests.
- Avoid a cross-package test utility package unless multiple packages need the same public fixture contract.
Suggested tests:
- This is test cleanup only; existing test assertions should remain equivalent.
Risk level: low.
## Boundary And Responsibility Concerns
- HTTP error mapping currently depends on runner error message text. Stable reason identity should live in `internal/usecase`; HTTP status/code/message mapping should remain in `internal/adapter/http`.
- Execution-target merge policy lives in `internal/usecase`, which fits the architecture. The concern is not placement but incomplete centralization of field mapping around that policy.
- CLI app wiring currently constructs concrete repositories/readers/renderers/validators inline in each command. That is acceptable adapter responsibility, but repeated wiring should be centralized within the CLI adapter package.
- Prompt/profile filesystem traversal is duplicated in two repository packages. A narrow filesystem catalog helper would fit the architecture because it would not own prompt/profile policy.
- LLM provider request shape is properly isolated in `internal/llm`. It should remain there; do not move OpenAI/OpenRouter request-field policy into runner or prompt definitions.
## Path, Key, And Naming Construction Review
Path and naming construction is mostly explicit and low-risk:
- Config paths are cleaned in `internal/config` and again in CLI finalization.
- Prompt `content_file` paths are resolved relative to the prompt YAML file in `internal/promptdef`.
- Schema paths are resolved through `internal/validate.StandardValidator`.
- OpenAI-compatible endpoint paths use `defaults.OpenAIChatCompletionsPath`.
- Output artifact name comes from `defaults.OutputArtifactName`.
Areas needing cleanup:
- Prompt/profile recursive YAML discovery and relative-path formatting should share one helper.
- CLI path cleaning after config resolution is repeated and should be consolidated with common command settings finalization.
- Structured schema names are derived in `internal/usecase`; this is currently one place and should stay there unless structured-output support expands.
No remote keys, cache paths, manifest paths, lock files, or generated report paths exist in the current implementation.
## Resolution And Catalog Review
Named concept resolution is mostly consistent:
- Prompt resolution uses YAML `id` and optional `version`, not file path.
- Profile resolution uses YAML `id`, not file path.
- Prompt/profile subdirectories are organizational only.
- Duplicate prompt/profile IDs fail instead of choosing first match.
- Schema resolution is explicit path-based relative to `schema_dir`; no basename search.
- Input artifacts are resolved by `artifact.Reader`, with `inline` and `file` as implemented types.
- Prompt required-input checks happen in the renderer.
Recommended centralization:
- Share only filesystem catalog mechanics between prompt/profile repositories.
- Keep prompt ID/version, profile ID, schema path, and input resolution policies in their current owning packages.
## Config And Command-Loading Review
Config precedence is consistent with policy:
1. built-in defaults;
2. config file values;
3. CLI overrides.
Intentional differences:
- `render` supports `--format`; `run` does not.
- `serve` supports `--addr`; `run`/`render` do not.
- `serve` rejects runtime model override flags.
- `render` does not expose `--schema-dir`, even though it can prepare `json_schema` prompts through config/default schema settings.
- CLI runtime model override flags are narrower than HTTP model override fields; this is currently documented by omission in CLI docs.
Likely cleanup areas:
- Common command settings finalization can be made smaller and less repetitive.
- Runner dependency construction should be a CLI-local helper.
- The `runConfig` flag-set booleans work, but adding more runtime override fields will continue to require touching several fields and the override condition.
## State, Manifest, Or Progress Handling Review
Scriptorium has no durable state, manifests, checkpoints, progress records, cache state, or resume behavior. This matches `docs/policy/architecture.md` and `docs/operations.md`.
There is no drift affecting resume, retry, force, dry-run, or audit behavior because those features do not exist. Validation repair hooks exist in the runner but are not wired by CLI/HTTP today; that boundary is documented.
## Refactors To Avoid
- Do not introduce a generic workflow/stage engine. Scriptorium intentionally executes one prompt request.
- Do not build a plugin architecture for prompt/profile/schema/artifact backends before another backend exists.
- Do not replace explicit CLI parse functions with a broad command framework.
- Do not merge CLI and HTTP adapters. Their public interfaces and failure semantics differ.
- Do not create a generic reflection-based mapper for domain/DTO/request structs.
- Do not centralize prompt/profile normalization into one generic YAML repository; their validation and error semantics are different.
- Do not add schema caching or a manifest/state store as part of deduplication.
- Do not document or implement unsupported artifact backends as cleanup.
## Recommended Implementation Sequence
1. Execution-target mapping cleanup
- Goal: reduce drift when adding profile/runtime/provider fields.
- Files to update: `internal/usecase`, `internal/adapter/http`, `internal/llm`, targeted tests.
- Acceptance criteria: all existing field mapping tests pass; a new all-fields test fails if a mapped execution field is omitted.
- Suggested validation: `go test ./internal/usecase ./internal/adapter/http ./internal/llm ./internal/format`.
- Small enough for one implementation prompt: yes.
2. Prompt/profile filesystem catalog helper
- Goal: centralize recursive YAML discovery, path sorting, extension filtering, and relative path formatting.
- Files to update: prompt/profile repositories plus new narrow helper package.
- Acceptance criteria: existing repository tests pass unchanged; new helper tests cover nested discovery and stable ordering.
- Suggested validation: `go test ./internal/promptdef ./internal/profile`.
- Small enough for one implementation prompt: yes.
3. CLI wiring and config finalization helper
- Goal: centralize common command settings resolution and runner construction.
- Files to update: `internal/adapter/cli/run.go` and CLI tests.
- Acceptance criteria: no flag behavior changes; run/render/serve config precedence tests pass.
- Suggested validation: `go test ./internal/adapter/cli`.
- Small enough for one implementation prompt: yes.
4. Stable use-case invalid-request reasons
- Goal: remove HTTP dependency on runner error message substrings.
- Files to update: `internal/usecase/runner.go`, `internal/adapter/http/handler.go`, tests.
- Acceptance criteria: HTTP error codes/messages remain unchanged; runner exposes stable `errors.Is` reason errors.
- Suggested validation: `go test ./internal/usecase ./internal/adapter/http`.
- Small enough for one implementation prompt: yes.
5. Test helper cleanup
- Goal: reduce repeated fixture construction after behavior-preserving refactors are complete.
- Files to update: primarily `internal/adapter/cli/run_test.go`, optionally `internal/usecase/runner_test.go`.
- Acceptance criteria: test intent remains clear; no cross-package helper package unless strongly justified.
- Suggested validation: `go test ./internal/adapter/cli ./internal/usecase`.
- Small enough for one implementation prompt: yes.
6. Dead-code/legacy sweep
- Goal: remove or clearly mark unsupported placeholders such as `ArtifactRefS3`.
- Files to update: domain/artifact tests/docs only if needed.
- Acceptance criteria: docs continue to describe only implemented behavior outside roadmap.
- Suggested validation: `go test ./internal/domain ./internal/artifact`.
- Small enough for one implementation prompt: yes.
## Test Strategy
Tests to add before or during cleanup:
- Execution target all-fields mapping tests:
- runner profile/default/request merge;
- HTTP request DTO to domain target;
- HTTP domain result to metadata DTO;
- LLM domain target to outbound JSON payload.
- Repository catalog helper tests:
- recursive scan;
- `.yaml` and `.yml` filtering;
- stable sorted paths;
- relative clean path formatting;
- context cancellation if the helper preserves current behavior.
- HTTP error mapping tests:
- profile-required and API-key-env-missing should rely on sentinel errors, not message text.
- CLI wiring regression tests:
- `run`, `render`, and `serve` preserve config precedence and required directory checks.
- Schema behavior tests:
- `Prepare` fails before LLM generation when `json_schema` structured-output schema cannot be loaded.
Validation commands for cleanup work:
- `go test ./internal/usecase ./internal/adapter/http ./internal/llm ./internal/format`
- `go test ./internal/promptdef ./internal/profile`
- `go test ./internal/adapter/cli`
- `go test ./...` before merging broader cleanup.
No automated docs or link checker is currently present.
## Appendix: Findings Not Worth Acting On
- Repeated `select { case <-ctx.Done(): ... }` checks are acceptable. They are local, simple, and appear at IO/loop boundaries where behavior is easy to read.
- CLI and HTTP request validation should remain separate. Their external contracts differ, and centralizing all request validation would blur adapter responsibilities.
- Prompt and profile validation should not be merged. They both use YAML and IDs, but their schemas, normalization rules, and error policies differ.
- Prepared-run text formatting is verbose but intentionally presentation-specific. Avoid abstracting it until another formatter needs the same layout policy.
- Hashing appears in artifact loading and runner metadata, but not all hashes represent the same thing. A tiny body-hash helper may be useful later; a generic hashing subsystem is not justified.
- HTTP DTOs duplicate domain field names by design. They should remain transport-owned so JSON compatibility can evolve deliberately.
- Config `applyConfig` and `ApplyCLIOverrides` look similar, but they apply different source labels and validation contexts. A broad merge abstraction would likely reduce clarity.

View File

@@ -1,504 +0,0 @@
# Cleanup Implementation Roadmap
## Purpose
This roadmap turns the findings in [audit.md](audit.md) into a staged, decision-complete cleanup plan for Scriptorium.
Audience: LLM coding agents implementing the cleanup in order.
Controlling policies:
- [Documentation policy](../policy/documentation.md)
- [Architecture policy](../policy/architecture.md)
- [Development guide](../policy/development.md)
## Global Implementation Rules
- Implement stages in order.
- Keep public CLI flags, HTTP request/response shapes, config precedence, prompt/profile ID semantics, and validation behavior stable unless a stage explicitly says otherwise.
- Keep adapters thin and use-case policy in `internal/usecase`.
- Prefer explicit helpers over reflection, generic workflow abstractions, or broad framework-style rewrites.
- Update non-roadmap docs only when implemented behavior changes or when a stale implemented-behavior statement is found during a stage.
- Do not document future behavior outside `docs/roadmap/`.
- Run the stage-specific tests before moving to the next stage.
- Run `go test ./...` after the final stage.
## Stage 1: Execution Target Mapping Cleanup
### Goal
Reduce drift when adding or changing execution/runtime fields such as `service_tier`, `api_key_env`, `reasoning_effort`, or future provider request keys.
### Scope
Update only explicit execution-target mapping and serialization paths. Do not add new CLI flags or new provider features.
### Implementation
In `internal/usecase`:
- Keep execution-target merge policy in `internal/usecase`.
- Add focused helper coverage around `resolveExecutionTarget`, `mergeExecutionTarget`, and profile-to-target conversion.
- Keep the existing semantics:
- built-in defaults first;
- profile values override defaults;
- request overrides override profile values;
- zero numeric values do not override;
- empty/whitespace string values do not override;
- non-empty `ExtraParams` replaces the previous map with a copy.
In `internal/adapter/http`:
- Add local helper functions:
- `executionTargetFromModelOverrideDTO(*modelOverrideRequestDTO) *domain.ExecutionTarget`
- `modelParamsDTOFromExecutionTarget(domain.ExecutionTarget) modelParamsDTO`
- Use those helpers in `handler.go`.
- Keep DTO types unexported and transport-owned.
- Keep HTTP response field names and omission behavior unchanged.
In `internal/llm`:
- Add a local helper such as `openAIChatRequestFromGenerateRequest(req domain.GenerateRequest, defaultModel string) (openAIChatRequest, error)` or an equivalent small function.
- Keep endpoint construction, HTTP client timeout handling, API-key environment lookup, and response parsing in `Generate`.
- Keep outbound serialization behavior unchanged:
- send `model` and `messages`;
- send `temperature`, `max_tokens`, `top_p`, and `service_tier` only when currently sent;
- send `response_format` only when structured output is present;
- do not serialize `reasoning_effort` or `extra_params`.
Do not:
- use reflection to copy fields;
- move HTTP DTOs into `internal/domain`;
- add generic mapper packages;
- change prepared-run JSON tags.
### Tests
Add or update tests so an all-fields `domain.ExecutionTarget` catches omissions.
Required tests:
- Runner merge/profile conversion:
- profile values populate all supported execution fields;
- runtime overrides beat profile values for all overrideable fields;
- empty string overrides do not erase profile values;
- empty `ExtraParams` does not erase profile values.
- HTTP adapter:
- request `model` object maps every supported field into `RunRequest.Execution`;
- response `metadata.model_params` includes every supported field according to current DTO tags.
- LLM client:
- outbound JSON includes every serialized execution field;
- outbound JSON omits `service_tier` when empty;
- outbound JSON still omits `reasoning_effort` and `extra_params`.
### Validation
Run:
```bash
go test ./internal/usecase ./internal/adapter/http ./internal/llm ./internal/format
```
### Acceptance Criteria
- No public behavior changes.
- Adding a new execution target field later has obvious mapping/test locations.
- Existing HTTP and LLM behavior remains stable.
- Stage is small enough for one implementation prompt.
## Stage 2: Prompt/Profile Filesystem Catalog Helper
### Goal
Centralize shared recursive YAML discovery mechanics while preserving prompt/profile-specific validation and error behavior.
### Scope
Create a narrow helper package for filesystem catalog mechanics only.
Recommended package:
- `internal/filecatalog`
### Implementation
Add helper functions with explicit, small responsibilities:
- recursively find YAML files under a root directory;
- honor context cancellation during walking;
- accept `.yaml` and `.yml`;
- return stable sorted full paths;
- compute clean relative paths from a root;
- return filename stems with `.yaml`/`.yml` stripped.
Use the helper in:
- `internal/promptdef/filesystem_repository.go`
- `internal/profile/filesystem_repository.go`
Preserve existing behavior:
- prompt/profile lookup uses YAML `id`, not file path;
- subdirectories are organizational only;
- duplicate prompt/profile IDs are invalid;
- malformed likely-target files still surface errors;
- relative nested paths still appear in errors;
- prompt `content_file` resolution remains relative to the prompt YAML file;
- prompt/profile strict YAML and validation stay in their existing packages.
Do not:
- create a generic repository framework;
- merge prompt and profile normalization;
- move prompt/profile domain policy into the helper;
- change error messages except for unavoidable wording caused by helper extraction.
### Tests
Add tests for `internal/filecatalog`:
- nested YAML discovery;
- `.yaml` and `.yml` accepted;
- non-YAML files ignored;
- returned paths sorted deterministically;
- relative path formatting works for nested files;
- filename stem stripping handles both extensions.
Keep existing prompt/profile repository tests passing.
### Validation
Run:
```bash
go test ./internal/filecatalog ./internal/promptdef ./internal/profile
```
### Acceptance Criteria
- Prompt/profile repository tests pass without behavior expectation changes.
- Shared filesystem scanning logic exists in one place.
- Prompt/profile packages still own their own validation and normalization.
- Stage is small enough for one implementation prompt.
## Stage 3: CLI Wiring And Settings Finalization Cleanup
### Goal
Reduce duplicated command setup while preserving each command's public flag surface and behavior.
### Scope
Clean up `internal/adapter/cli` only, except for tests.
### Implementation
Add CLI-local helpers. Recommended helpers:
- `commonCommandSettings` or similar struct containing resolved `promptDir`, `profileDir`, `schemaDir`, `serverAddr`, and `defaultRenderFormat` where applicable.
- `resolveCommonSettings(fs *flag.FlagSet, configPath string, overrides appconfig.CLIOverrides) (commonCommandSettings, error)`.
- `validateRequiredLibraryDirs(promptDir, profileDir string) error`.
- `newRunner(promptDir, profileDir, schemaDir string, llmClient llm.Client) *usecase.Runner`.
- Optionally `newOpenAIClient() (*llm.OpenAICompatibleClient, error)` if it removes exact duplication without obscuring command behavior.
Preserve command differences:
- `run` exposes runtime model override flags and `--schema-dir`;
- `render` exposes runtime model override flags and `--format`, but not `--schema-dir`;
- `serve` exposes `--addr` and `--schema-dir`, but no runtime model override flags;
- `render` default format comes from `defaults.render_format` unless `--format` is set;
- deprecated `--prompt-id` and `--profile-id` aliases remain accepted.
Keep existing parse functions:
- `parseRunArgs`
- `parseRenderArgs`
- `parseServeArgs`
Do not:
- replace the standard library `flag` package;
- introduce a command framework;
- make `serve` accept runtime model override flags;
- change error prefixes such as `run parse error`, `render parse error`, or `serve parse error`;
- change CLI output behavior.
### Tests
Required regression tests:
- `run`, `render`, and `serve` still apply config precedence correctly.
- Missing effective `prompt_dir` and `profile_dir` still return the same guidance.
- `render` still uses config default render format and explicit `--format` override.
- `serve` still rejects runtime model override flags.
- `run` and `render` still build equivalent runtime override requests for shared flags.
### Validation
Run:
```bash
go test ./internal/adapter/cli
```
### Acceptance Criteria
- No CLI flag, output, exit-code, or precedence changes.
- Runner dependency construction is centralized inside the CLI adapter.
- Command-specific behavior remains easy to read.
- Stage is small enough for one implementation prompt.
## Stage 4: Stable Use-Case Error Reasons For HTTP Mapping
### Goal
Remove HTTP error mapping's dependency on runner error message substrings.
### Scope
Change error identity, not public HTTP error responses.
### Implementation
In `internal/usecase`:
- Add stable sentinel errors for invalid-request reasons that HTTP currently distinguishes by message text.
- Required sentinels:
- missing profile selection, for the case where neither request profile nor prompt `default_profile` is available;
- missing API-key environment value, for the case where `api_key_env` is set but the named environment variable is unset or empty.
- Wrap these sentinels with `ErrInvalidRequest` so existing broad invalid-request checks keep working.
- Preserve clear human-readable runner errors.
In `internal/adapter/http`:
- Replace `strings.Contains(err.Error(), ...)` checks for these cases with `errors.Is`.
- Keep current HTTP status codes, error codes, and response messages:
- `400 profile_required`;
- `400 api_key_env_missing`.
Do not:
- expose HTTP-specific error codes from `internal/usecase`;
- change the HTTP JSON error body shape;
- remove broad fallback handling for `usecase.ErrInvalidRequest`.
### Tests
Required tests:
- Runner tests assert `errors.Is(err, usecase.ErrProfileRequired)` or the chosen sentinel name for missing profile selection.
- Runner tests assert `errors.Is(err, usecase.ErrAPIKeyEnvMissing)` or the chosen sentinel name for missing API-key environment value.
- HTTP handler tests still assert unchanged status/code/message for both cases.
- HTTP handler tests should not construct errors by relying on exact runner prose for these two cases.
### Validation
Run:
```bash
go test ./internal/usecase ./internal/adapter/http
```
### Acceptance Criteria
- HTTP mapping no longer depends on runner message substrings for the two distinguished invalid-request cases.
- Public HTTP behavior is unchanged.
- Runner errors remain clear in CLI output.
- Stage is small enough for one implementation prompt.
## Stage 5: Schema Failure Regression Coverage
### Goal
Protect the current structured-output invariant before future schema cleanup: `Prepare` must load a `json_schema` document before any LLM call.
### Scope
Add regression coverage only. Do not add schema caching or change schema loading architecture in this stage.
### Implementation
In `internal/usecase/runner_test.go` or an appropriate package test:
- Add a test where a prompt uses `validation_mode: json_schema` with a missing or failing schema document.
- Assert `Runner.Prepare` fails with `ErrValidation`.
- Assert no LLM call is made for `Runner.Run` when structured-output schema loading fails.
If existing tests already cover part of this behavior, consolidate assertions without making the test suite harder to read.
Do not:
- cache compiled schemas;
- change `validate.StandardValidator` behavior;
- introduce a schema service abstraction.
### Validation
Run:
```bash
go test ./internal/usecase ./internal/validate
```
### Acceptance Criteria
- Missing structured-output schema fails before generation.
- Existing JSON Schema validation behavior is unchanged.
- Stage is small enough for one implementation prompt.
## Stage 6: Test Fixture Cleanup
### Goal
Reduce repeated test setup after behavior-preserving production refactors are complete.
### Scope
Prefer package-local test helpers. Avoid cross-package test utility packages unless a helper is needed by more than two packages and represents a stable public fixture contract.
### Implementation
In `internal/adapter/cli/run_test.go`:
- Consolidate repeated temp prompt/profile/input setup into local helper functions.
- Keep helper names behavior-focused, for example:
- `newCLITestLibrary`
- `writePromptFileWithDefaultProfile`
- `writeProfileFile`
- `runCLICommand`
- Do not hide assertions inside helpers unless the assertion is truly setup validation.
In `internal/usecase/runner_test.go`:
- Keep existing fake interfaces package-local.
- Remove only high-volume duplication that obscures test intent.
Do not:
- move package-private fake types into production code;
- create a broad `internal/testutil` package unless a later cleanup stage proves it necessary;
- rewrite tests into table-driven form when cases have meaningfully different setup.
### Validation
Run:
```bash
go test ./internal/adapter/cli ./internal/usecase
```
### Acceptance Criteria
- Test intent is at least as clear as before.
- No production behavior changes.
- Test fixture setup has less repeated boilerplate in CLI tests.
- Stage is small enough for one implementation prompt.
## Stage 7: Unsupported Placeholder Sweep
### Goal
Remove code that suggests unimplemented artifact behavior outside roadmap documentation.
### Scope
Remove unsupported placeholders only when they are not needed by current tests or public docs.
### Implementation
Remove `domain.ArtifactRefS3` from `internal/domain/domain.go` unless new evidence shows it is intentionally needed by implemented code.
Preserve current behavior:
- supported artifact reference types remain `inline` and `file`;
- unsupported artifact reference types still return `artifact.ErrUnsupportedRefType`;
- docs continue to describe only `inline` and `file` outside roadmap files.
Update tests only if they reference the removed constant. Prefer testing unsupported artifact behavior with a literal custom type such as `domain.ArtifactRefType("s3")` or `domain.ArtifactRefType("unsupported")`.
Do not:
- add S3 support;
- document S3 as implemented;
- add future-backend placeholders elsewhere.
### Validation
Run:
```bash
go test ./internal/domain ./internal/artifact
```
### Acceptance Criteria
- Unsupported placeholder constant is removed.
- Unsupported artifact-type behavior remains covered.
- No non-roadmap doc claims unimplemented artifact support.
- Stage is small enough for one implementation prompt.
## Stage 8: Final Verification And Documentation Alignment
### Goal
Confirm the cleanup sequence preserved behavior and documentation accuracy.
### Implementation
Run the full test suite:
```bash
go test ./...
```
Run the maintained render smoke command:
```bash
go run ./cmd/scriptorium render --config ./examples/config.yml --prompt generic.markdown_summary --input transcript=./examples/fixtures/transcript.md --input glossary=./examples/fixtures/glossary.yml --format json
```
Search for stale or unsupported terms:
```bash
rg -n "ArtifactRefS3|s3|strings\\.Contains\\(err\\.Error\\(\\)|TODO|future|planned" internal docs README.md examples
```
Review results manually:
- `s3` should not appear as an implemented artifact type.
- `strings.Contains(err.Error())` should not be used for stable use-case reason mapping.
- Any `future` or `planned` wording outside `docs/roadmap/` must describe current boundaries, not aspirational behavior.
Update docs only if cleanup changed implemented behavior or if the search reveals stale implemented-behavior docs.
### Acceptance Criteria
- Full test suite passes.
- Maintained render smoke command succeeds.
- No stale unsupported artifact placeholder remains.
- Non-roadmap docs describe implemented behavior only.
- Working tree contains only intentional cleanup changes.
## Deferred Work
Do not implement these during the staged cleanup unless a later audit makes them high-confidence:
- schema caching or a combined raw/compiled schema service;
- artifact hash/build helper beyond a small helper introduced opportunistically during touched code;
- generic YAML repository framework;
- generic CLI command framework;
- plugin architecture for future prompt/profile/schema/artifact backends;
- durable state, manifests, checkpoints, or resume behavior.
## Completion Criteria
The cleanup roadmap is complete when all stages have been implemented in order, the final verification passes, and the resulting code still satisfies:
- `Runner.Run` reuses `Runner.Prepare`;
- CLI and HTTP adapters instantiate `Runner` without a repairer;
- unknown config/prompt/profile YAML and HTTP JSON fields are rejected;
- raw API key values are not accepted or emitted;
- prompt/profile subdirectories remain organizational only;
- schema paths remain explicit and relative to `schema_dir` when not absolute;
- public CLI and HTTP behavior remains stable.

View File

@@ -1,19 +1,25 @@
# Troubleshooting # Troubleshooting
This guide lists recurring implemented failure modes and safe fixes. This guide lists common implemented failure modes and safe fixes.
For command syntax, see [CLI reference](cli.md). For configuration and file formats, see [Configuration reference](config.md). For operational behavior, see [Operations guide](operations.md). Canonical references:
## Missing Or Invalid Config File - [CLI reference](cli.md)
- [Configuration reference](config.md)
- [HTTP API reference](api.md)
- [Operations guide](operations.md)
## Missing Or Invalid Config
Symptom: Symptom:
- CLI errors such as `application config error: config file not found` or `invalid config YAML`. - CLI error includes `application config error`, `config file not found`, `invalid config YAML`, or `invalid config`.
Likely cause: Likely cause:
- `--config` points to a missing file. - `--config` points to a missing file.
- Config YAML has syntax errors or unknown fields. - YAML syntax is invalid.
- Config contains unknown fields or negative HTTP size limits.
Diagnostic step: Diagnostic step:
@@ -23,40 +29,34 @@ go run ./cmd/scriptorium render --config /path/to/config.yml --prompt generic.ma
Safe fix: Safe fix:
- Correct file path. - Correct the config path.
- Remove unknown fields.
- Fix YAML syntax. - Fix YAML syntax.
- Keep secrets out of config. - Remove unknown fields.
- Keep raw secrets out of config.
Relevant links: Relevant links: [Configuration reference](config.md), [CLI reference](cli.md)
- [Configuration reference](config.md) ## Missing Prompt Directory
- [CLI reference](cli.md)
## Missing Prompt/Profile Directory Settings
Symptom: Symptom:
- CLI parse errors saying prompt directory or profile directory is required. - CLI parse error says the prompt directory is required.
Likely cause: Likely cause:
- Neither CLI flags nor config provide effective `prompt_dir` / `profile_dir`. - Neither config nor CLI flags provide an effective `prompt_dir`.
Diagnostic step: Diagnostic step:
- Run the failing command with explicit `--prompt-dir` and `--profile-dir` once to verify. - Re-run once with explicit `--prompt-dir`.
Safe fix: Safe fix:
- Set `prompt_dir` and `profile_dir` in config, or always pass both flags. - Set `prompt_dir` in config or pass `--prompt-dir`.
Relevant links: Relevant links: [Configuration reference](config.md), [CLI reference](cli.md)
- [Configuration reference](config.md) ## Unknown Flags
- [CLI reference](cli.md)
## Unknown Or Unsupported Flags
Symptom: Symptom:
@@ -64,33 +64,33 @@ Symptom:
Likely cause: Likely cause:
- Typo or command mismatch (for example, `serve` with runtime model override flags). - Typo.
- Flag is valid for another command.
- `serve` was given runtime model override flags.
Diagnostic step: Diagnostic step:
- Compare command against the command-specific flag list. - Compare the command with the command-specific flag list.
Safe fix: Safe fix:
- Remove unsupported flags. - Remove unsupported flags.
- Use `run`/`render` for runtime model overrides. - Use `run` or `render` for runtime model overrides.
Relevant links: Relevant links: [CLI reference](cli.md)
- [CLI reference](cli.md) ## Prompt Load Failures
## Prompt Definition Load Failures
Symptom: Symptom:
- CLI run/render error from prompt loading. - CLI run/render fails during prompt loading.
- HTTP `404 prompt_not_found` or `400 prompt_load_failed`. - HTTP returns `404 prompt_not_found` or `400 prompt_load_failed`.
Likely cause: Likely cause:
- Prompt ID not found. - Prompt ID/version does not exist.
- Invalid prompt YAML. - Prompt YAML is invalid or has unknown fields.
- Invalid prompt contract (for example bad validation mode, message content/content_file rule violation, missing schema path for `json_schema`). - Prompt contract is invalid, such as missing messages, invalid output mode, bad `content_file`, or missing `schema_path` for `json_schema`.
Diagnostic step: Diagnostic step:
@@ -100,28 +100,25 @@ go run ./cmd/scriptorium render --config ./examples/config.yml --prompt <prompt-
Safe fix: Safe fix:
- Correct prompt ID. - Correct prompt ID/version.
- Fix prompt YAML and contract fields. - Fix prompt YAML and referenced `content_file` paths.
- Ensure referenced `content_file` paths exist. - Fix output contract fields.
Relevant links: Relevant links: [Configuration reference](config.md), [CLI reference](cli.md)
- [Configuration reference](config.md) ## Profile Load Failures
- [CLI reference](cli.md)
## Profile Definition Load Failures
Symptom: Symptom:
- CLI run/render error from profile loading. - CLI run/render fails during profile loading.
- HTTP `404 profile_not_found` or `400 profile_load_failed`. - HTTP returns `404 profile_not_found`, `400 profile_load_failed`, or `400 profile_required`.
Likely cause: Likely cause:
- Profile ID missing/not found. - Profile ID does not exist.
- Invalid profile YAML. - Request omitted profile and prompt has no `default_profile`.
- Invalid profile values. - Profile YAML is invalid or has unknown fields.
- Raw `api_key` field present (rejected). - Profile contains raw `api_key`.
Diagnostic step: Diagnostic step:
@@ -131,78 +128,52 @@ go run ./cmd/scriptorium render --config ./examples/config.yml --prompt generic.
Safe fix: Safe fix:
- Correct profile ID. - Correct profile ID or prompt `default_profile`.
- Fix profile YAML and value ranges. - Fix profile YAML and value ranges.
- Replace `api_key` with `api_key_env`. - Replace raw `api_key` with `api_key_env`.
Relevant links: Relevant links: [Configuration reference](config.md), [CLI reference](cli.md)
- [Configuration reference](config.md) ## Input Artifact Failures
- [CLI reference](cli.md)
## Input Artifact Read Failures
Symptom: Symptom:
- CLI run/render error reading input artifacts. - CLI run/render fails while reading inputs.
- HTTP `400 artifact_read_failed`. - HTTP returns `400 artifact_read_failed`, `400 artifact_not_allowed`, or `413 artifact_too_large`.
Likely cause: Likely cause:
- File path in input mapping does not exist or is unreadable. - Input file path is missing or unreadable.
- Unsupported artifact reference type in HTTP request. - HTTP input type is unsupported or missing required fields.
- HTTP file refs are disabled because no artifact root is configured.
- HTTP file path is lexically outside the artifact root.
- HTTP file input exceeds `server.max_artifact_bytes`.
Diagnostic step: Diagnostic step:
- Verify every mapped file path exists and is readable by the process. - Verify each input path exists and is readable by the process.
- For HTTP, verify each input uses supported `type` values. - For HTTP, verify input refs use `file` or `inline`.
- For HTTP file refs, verify the artifact root and compare file size to `server.max_artifact_bytes`.
Safe fix: Safe fix:
- Correct file paths and permissions. - Correct paths and permissions.
- Use supported input types (`file`, `inline`). - Configure a narrow artifact root for HTTP file refs.
- Use relative paths under the artifact root or switch to `inline`.
- Increase `server.max_artifact_bytes` only for expected larger inputs.
Relevant links: Relevant links: [HTTP API reference](api.md), [Configuration reference](config.md)
- [CLI reference](cli.md)
- [Configuration reference](config.md)
## Prompt Template Render Failures
Symptom:
- CLI run/render error from prompt rendering.
- HTTP `400 prompt_render_failed`.
Likely cause:
- Template references missing input names.
- Template syntax or data reference issues.
Diagnostic step:
- Run `render --format json` with the same prompt, inputs, vars, and profile selection.
Safe fix:
- Align template `{{input "name"}}` references with actual input mappings.
- Fix template syntax and variable names.
Relevant links:
- [CLI reference](cli.md)
- [Configuration reference](config.md)
## Missing API-Key Environment Variable ## Missing API-Key Environment Variable
Symptom: Symptom:
- CLI run/render invalid request error about missing API-key environment variable. - CLI render/run fails with an API-key environment error.
- HTTP `400 api_key_env_missing`. - HTTP returns `400 api_key_env_missing`.
Likely cause: Likely cause:
- Selected profile or override sets `api_key_env`, but that environment variable is unset/empty. - Selected profile or runtime override sets `api_key_env`, but the environment variable is unset or empty.
Diagnostic step: Diagnostic step:
@@ -212,52 +183,68 @@ printenv SCRIPTORIUM_API_KEY
Safe fix: Safe fix:
- Set the required environment variable before invoking CLI/service. - Set the required environment variable before starting the CLI command or HTTP service.
- Or use a profile that does not require API key auth for the target endpoint. - Or use a profile that does not require provider API-key auth.
Relevant links: Relevant links: [Configuration reference](config.md), [Operations guide](operations.md)
- [Configuration reference](config.md) ## Prompt Template Render Failures
- [Operations guide](operations.md)
Symptom:
- CLI render/run fails during prompt rendering.
- HTTP returns `400 prompt_render_failed`.
Likely cause:
- Template references an input that was not supplied.
- Template syntax or variable reference is invalid.
Diagnostic step:
- Run `render --format json` with the same prompt, inputs, vars, and profile.
Safe fix:
- Align `{{input "name"}}` references with request input names.
- Fix template syntax and variable names.
Relevant links: [Configuration reference](config.md), [CLI reference](cli.md)
## LLM Request Failures ## LLM Request Failures
Symptom: Symptom:
- CLI `run` fails with LLM generation errors. - CLI `run` fails during generation.
- HTTP returns `502 llm_failed`. - HTTP returns `502 llm_failed`.
Likely cause: Likely cause:
- Endpoint unreachable. - Endpoint is unreachable.
- Non-2xx response from provider. - Provider returns non-2xx.
- Timeout. - Request times out.
- Malformed provider response. - Provider response is malformed.
Diagnostic step: Diagnostic step:
- Confirm endpoint URL and model in selected profile/overrides. - Run `render` first to confirm pre-LLM preparation works.
- Retry with `render` first to confirm pre-LLM preparation works. - Check selected endpoint/model in prepared output.
- Check provider/network logs for non-2xx responses and timeouts. - Check network/provider logs for timeout or non-2xx details.
Safe fix: Safe fix:
- Correct endpoint/model settings. - Correct endpoint/model/profile settings.
- Adjust timeout if needed. - Adjust timeout when appropriate.
- Resolve provider-side or network issues. - Resolve provider or network issue.
Relevant links: Relevant links: [Operations guide](operations.md), [Configuration reference](config.md)
- [CLI reference](cli.md) ## Validation Failed
- [Configuration reference](config.md)
- [Operations guide](operations.md)
## Validation Status Failed (`run` Exit 2 Or HTTP 200 With Failed Status)
Symptom: Symptom:
- CLI exits with code `2`. - CLI `run` exits `2`.
- HTTP returns `200`, but `validation.status` is `failed`. - HTTP returns `200 OK` with `validation.status` set to `failed`.
Likely cause: Likely cause:
@@ -265,18 +252,15 @@ Likely cause:
Diagnostic step: Diagnostic step:
- Inspect validation mode and validation errors in CLI summary/HTTP response. - Inspect validation errors in CLI stderr or the HTTP response.
Safe fix: Safe fix:
- Refine prompt constraints. - Refine prompt instructions.
- Tighten schema or adjust model/profile settings. - Adjust schema or model/profile settings.
- Rerun after correction. - Rerun after correction.
Relevant links: Relevant links: [Operations guide](operations.md), [HTTP API reference](api.md)
- [Configuration reference](config.md)
- [Operations guide](operations.md)
## Validation Runtime Failure ## Validation Runtime Failure
@@ -287,48 +271,91 @@ Symptom:
Likely cause: Likely cause:
- `json_schema` schema file missing/inaccessible. - `json_schema` schema file is missing or unreadable.
- Invalid schema JSON document. - Schema JSON is invalid.
Diagnostic step: Diagnostic step:
- Verify `schema_dir` and `output.schema_path` resolution. - Verify `schema_dir` and prompt `output.schema_path`.
- Check schema file readability and valid JSON syntax. - Check schema file readability and JSON syntax.
Safe fix: Safe fix:
- Correct schema path. - Correct schema path or permissions.
- Fix schema JSON content. - Fix schema JSON.
- Rerun. - Rerun.
Relevant links: Relevant links: [Configuration reference](config.md), [Operations guide](operations.md)
- [Configuration reference](config.md) ## HTTP JSON Or Request Contract Errors
- [Operations guide](operations.md)
## HTTP Request Parsing/Contract Errors
Symptom: Symptom:
- HTTP `400 invalid_json` or `400 invalid_request`. - HTTP returns `400 invalid_json` or `400 invalid_request`.
Likely cause: Likely cause:
- Malformed JSON body. - JSON body is malformed.
- Unknown JSON fields. - Request has unknown fields or trailing JSON tokens.
- Missing required `prompt_id` or `inputs`. - Required `prompt_id` or `inputs` is missing.
- Runtime override values are out of range.
- `extra_params` collides with reserved outbound fields.
Diagnostic step: Diagnostic step:
- Revalidate request JSON. - Revalidate request JSON and compare fields with the API reference.
- Confirm required request fields are present.
Safe fix: Safe fix:
- Send valid JSON with only supported fields. - Send one JSON object with only supported fields.
- Ensure `prompt_id` and at least one input mapping are included. - Include `prompt_id` and at least one input.
- Use valid model override ranges.
- Remove reserved `extra_params` keys.
Relevant links: Relevant links: [HTTP API reference](api.md)
- [Operations guide](operations.md) ## HTTP Size Limit Errors
- [CLI reference](cli.md)
Symptom:
- HTTP returns `413 request_too_large`, `413 artifact_too_large`, or `413 response_too_large`.
Likely cause:
- JSON request body exceeds `server.max_request_bytes`.
- HTTP file input exceeds `server.max_artifact_bytes`.
- Encoded JSON response exceeds `server.max_response_bytes`.
Diagnostic step:
- Compare request, file input, and expected response sizes with configured limits.
Safe fix:
- Use smaller inline inputs or switch to file inputs under the artifact root.
- Reduce generated output size.
- Omit `include_raw_output`.
- Increase limits only when the deployment expects larger payloads.
Relevant links: [HTTP API reference](api.md), [Operations guide](operations.md)
## HTTP Route Or Method Errors
Symptom:
- HTTP returns `404 not_found` or `405 method_not_allowed`.
Likely cause:
- Path is not `/v1/runs`.
- Method on `/v1/runs` is not `POST`.
Diagnostic step:
- Check the request URL and method.
Safe fix:
- Send `POST /v1/runs`.
Relevant links: [HTTP API reference](api.md)

318
engine.go Normal file
View File

@@ -0,0 +1,318 @@
package scriptorium
import (
"context"
"errors"
"fmt"
"io/fs"
"net/http"
"os"
"path/filepath"
"strings"
"time"
artifactadapter "gitea.maximumdirect.net/eric/scriptorium/internal/artifact"
"gitea.maximumdirect.net/eric/scriptorium/internal/defaults"
"gitea.maximumdirect.net/eric/scriptorium/internal/llm"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile/builtin"
"gitea.maximumdirect.net/eric/scriptorium/internal/prompt"
"gitea.maximumdirect.net/eric/scriptorium/internal/promptdef"
"gitea.maximumdirect.net/eric/scriptorium/internal/usecase"
"gitea.maximumdirect.net/eric/scriptorium/internal/validate"
)
// ErrInvalidConfig indicates invalid public engine configuration.
var ErrInvalidConfig = errors.New("invalid engine configuration")
var (
ErrInvalidRequest = errors.New("invalid run request")
ErrPromptNotFound = errors.New("prompt not found")
ErrProfileNotFound = errors.New("profile not found")
ErrPromptLoad = errors.New("failed to load prompt definition")
ErrProfileLoad = errors.New("failed to load execution profile")
ErrArtifactLoad = errors.New("failed to load artifact")
ErrPromptRender = errors.New("failed to render prompt")
ErrLLMGenerate = errors.New("failed to generate output")
ErrValidation = errors.New("failed to validate output")
)
// Engine prepares and runs Scriptorium prompt requests.
type Engine struct {
runner *usecase.Runner
}
// Config configures a public Scriptorium engine.
type Config struct {
PromptDir string
ProfileDir string
SchemaDir string
Timeout time.Duration
HTTPClient *http.Client
}
// Option customizes engine construction.
type Option interface {
apply(*engineOptions) error
}
type optionFunc func(*engineOptions) error
func (f optionFunc) apply(options *engineOptions) error {
return f(options)
}
type engineOptions struct {
llmClient llm.Client
promptDefs promptdef.Repository
profiles profile.Repository
memoryProfiles profile.Repository
validator validate.Validator
promptSource bool
profileSource bool
memorySource bool
validatorSource bool
}
// WithLLMClient injects a custom LLM client for execution.
func WithLLMClient(client LLMClient) Option {
return optionFunc(func(options *engineOptions) error {
if client == nil {
return ErrInvalidConfig
}
options.llmClient = publicLLMClientAdapter{client: client}
return nil
})
}
// WithPromptFS loads prompt definitions from fsys under root.
//
// The source uses the same strict prompt YAML rules as configured prompt
// directories, and prompt content_file paths resolve within this source.
func WithPromptFS(fsys fs.FS, root string) Option {
return optionFunc(func(options *engineOptions) error {
if fsys == nil {
return ErrInvalidConfig
}
if strings.TrimSpace(root) == "" {
return ErrInvalidConfig
}
options.promptDefs = promptdef.NewFSRepository(fsys, root)
options.promptSource = true
return nil
})
}
// WithPromptFile loads prompt definitions from the single prompt file at path.
//
// Relative prompt content_file paths resolve from the file's directory.
func WithPromptFile(path string) Option {
return optionFunc(func(options *engineOptions) error {
fsys, root, err := fileSource(path)
if err != nil {
return err
}
options.promptDefs = promptdef.NewFSRepository(fsys, root)
options.promptSource = true
return nil
})
}
// WithProfileFS loads execution profiles from fsys under root.
//
// Profiles from this source overlay built-in profiles. Profile YAML must use
// api_key_env for environment-based credentials; raw API keys are rejected.
func WithProfileFS(fsys fs.FS, root string) Option {
return optionFunc(func(options *engineOptions) error {
if fsys == nil {
return ErrInvalidConfig
}
if strings.TrimSpace(root) == "" {
return ErrInvalidConfig
}
options.profiles = profile.NewFSRepository(fsys, root)
options.profileSource = true
return nil
})
}
// WithProfileFile loads execution profiles from the single profile file at path.
//
// The profile overlays built-in profiles. Profile YAML must use api_key_env for
// environment-based credentials; raw API keys are rejected.
func WithProfileFile(path string) Option {
return optionFunc(func(options *engineOptions) error {
fsys, root, err := fileSource(path)
if err != nil {
return err
}
options.profiles = profile.NewFSRepository(fsys, root)
options.profileSource = true
return nil
})
}
// WithProfiles configures in-memory profiles that take precedence over
// configured profile files and built-in profiles.
func WithProfiles(profiles ...Profile) Option {
return optionFunc(func(options *engineOptions) error {
repo, err := newMemoryProfileRepository(profiles)
if err != nil {
return err
}
options.memoryProfiles = repo
options.memorySource = true
return nil
})
}
// WithSchemaFS loads JSON Schema documents from fsys under root.
//
// Prompt schema_path values resolve within this source when schema validation
// or structured output is requested.
func WithSchemaFS(fsys fs.FS, root string) Option {
return optionFunc(func(options *engineOptions) error {
if fsys == nil {
return ErrInvalidConfig
}
if strings.TrimSpace(root) == "" {
return ErrInvalidConfig
}
options.validator = validate.NewFSValidator(fsys, root)
options.validatorSource = true
return nil
})
}
// WithSchemaFile loads JSON Schema documents from the single schema file at path.
//
// Prompt schema_path values refer to the file's base name.
func WithSchemaFile(path string) Option {
return optionFunc(func(options *engineOptions) error {
fsys, root, err := fileSource(path)
if err != nil {
return err
}
options.validator = validate.NewFSValidator(fsys, root)
options.validatorSource = true
return nil
})
}
// NewEngine constructs an Engine using the same default internal components as
// the CLI and HTTP adapters.
func NewEngine(cfg Config, opts ...Option) (*Engine, error) {
var options engineOptions
for _, opt := range opts {
if opt == nil {
continue
}
if err := opt.apply(&options); err != nil {
return nil, fmt.Errorf("%w: %v", ErrInvalidConfig, err)
}
}
promptDefs := options.promptDefs
if !options.promptSource {
if strings.TrimSpace(cfg.PromptDir) == "" {
return nil, fmt.Errorf("%w: prompt directory is required", ErrInvalidConfig)
}
promptDefs = promptdef.NewFilesystemRepository(cfg.PromptDir)
}
profiles := builtin.NewRepositoryWithDirectory(cfg.ProfileDir)
if options.profileSource {
profiles = builtin.NewRepositoryWithPrimary(options.profiles)
}
if options.memorySource {
profiles = profile.NewOverlayRepository(options.memoryProfiles, profiles)
}
validator := options.validator
if !options.validatorSource {
schemaDir := cfg.SchemaDir
if strings.TrimSpace(schemaDir) == "" {
schemaDir = defaults.SchemaDirDefault
}
validator = validate.NewStandardValidator(schemaDir)
}
llmClient := options.llmClient
if llmClient == nil {
var err error
llmClient, err = llm.NewOpenAICompatibleClient(llm.OpenAICompatibleConfig{
Timeout: cfg.Timeout,
HTTPClient: cfg.HTTPClient,
})
if err != nil {
return nil, fmt.Errorf("%w: %v", ErrInvalidConfig, err)
}
}
return &Engine{
runner: usecase.NewRunner(
promptDefs,
profiles,
artifactadapter.NewCompositeReader(),
prompt.NewGoRenderer(),
llmClient,
validator,
),
}, nil
}
func fileSource(name string) (fs.FS, string, error) {
cleanName := strings.TrimSpace(name)
if cleanName == "" {
return nil, "", ErrInvalidConfig
}
dir := filepath.Dir(cleanName)
base := filepath.Base(cleanName)
if base == "." || base == string(filepath.Separator) || strings.TrimSpace(base) == "" {
return nil, "", ErrInvalidConfig
}
info, err := os.Stat(cleanName)
if err != nil {
return nil, "", fmt.Errorf("%w: failed to access source file %q: %v", ErrInvalidConfig, cleanName, err)
}
if info.IsDir() {
return nil, "", fmt.Errorf("%w: source path %q must be a file", ErrInvalidConfig, cleanName)
}
return os.DirFS(dir), filepath.ToSlash(base), nil
}
// Prepare resolves a prompt request without calling an LLM.
func (e *Engine) Prepare(ctx context.Context, req RunRequest) (*PreparedRun, error) {
if e == nil || e.runner == nil {
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
}
domainReq, err := toDomainRunRequest(req)
if err != nil {
return nil, fmt.Errorf("%w: %v", ErrInvalidRequest, err)
}
prepared, err := e.runner.Prepare(ctx, domainReq)
if err != nil {
return nil, mapPublicError(err)
}
return fromDomainPreparedRun(prepared), nil
}
// Run executes a prompt request and returns the generated artifact and metadata.
func (e *Engine) Run(ctx context.Context, req RunRequest) (*RunResult, error) {
if e == nil || e.runner == nil {
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
}
domainReq, err := toDomainRunRequest(req)
if err != nil {
return nil, fmt.Errorf("%w: %v", ErrInvalidRequest, err)
}
result, err := e.runner.Run(ctx, domainReq)
if err != nil {
return nil, mapPublicError(err)
}
return fromDomainRunResult(result), nil
}

1799
engine_test.go Normal file

File diff suppressed because it is too large Load Diff

79
errors.go Normal file
View File

@@ -0,0 +1,79 @@
package scriptorium
import (
"errors"
"fmt"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile"
"gitea.maximumdirect.net/eric/scriptorium/internal/promptdef"
"gitea.maximumdirect.net/eric/scriptorium/internal/usecase"
)
func mapPublicError(err error) error {
if err == nil {
return nil
}
if hasPublicError(err) {
return err
}
publicErr := publicErrorFor(err)
if publicErr == nil {
return err
}
return fmt.Errorf("%w: %w", publicErr, err)
}
func hasPublicError(err error) bool {
for _, publicErr := range []error{
ErrInvalidConfig,
ErrInvalidRequest,
ErrPromptNotFound,
ErrProfileNotFound,
ErrPromptLoad,
ErrProfileLoad,
ErrArtifactLoad,
ErrPromptRender,
ErrLLMGenerate,
ErrValidation,
} {
if errors.Is(err, publicErr) {
return true
}
}
return false
}
func publicErrorFor(err error) error {
switch {
case errors.Is(err, promptdef.ErrPromptDefinitionNotFound):
return ErrPromptNotFound
case errors.Is(err, profile.ErrProfileNotFound):
return ErrProfileNotFound
case errors.Is(err, usecase.ErrPromptLoad):
return ErrPromptLoad
case errors.Is(err, usecase.ErrProfileLoad):
return ErrProfileLoad
case errors.Is(err, promptdef.ErrInvalidYAML), errors.Is(err, promptdef.ErrInvalidPromptDefinition):
return ErrPromptLoad
case isProfileLoadCause(err):
return ErrProfileLoad
case errors.Is(err, usecase.ErrArtifactLoad):
return ErrArtifactLoad
case errors.Is(err, usecase.ErrPromptRender):
return ErrPromptRender
case errors.Is(err, usecase.ErrLLMGenerate):
return ErrLLMGenerate
case errors.Is(err, usecase.ErrValidation):
return ErrValidation
case errors.Is(err, usecase.ErrInvalidRequest):
return ErrInvalidRequest
default:
return nil
}
}
func isProfileLoadCause(err error) bool {
return errors.Is(err, profile.ErrInvalidYAML) ||
errors.Is(err, profile.ErrInvalidProfile) ||
errors.Is(err, profile.ErrRawAPIKeyNotAllowed)
}

13
examples/config.full.yml Normal file
View File

@@ -0,0 +1,13 @@
prompt_dir: ./examples/prompts
profile_dir: ./examples/profiles
schema_dir: ./examples/schemas
server:
addr: 127.0.0.1:8080
artifact_root: .
max_request_bytes: 16777216
max_artifact_bytes: 16777216
max_response_bytes: 16777216
defaults:
render_format: text

View File

@@ -4,6 +4,7 @@ schema_dir: ./examples/schemas
server: server:
addr: :8080 addr: :8080
artifact_root: .
defaults: defaults:
render_format: text render_format: text

View File

@@ -0,0 +1,50 @@
package main
import (
"context"
"encoding/json"
"log"
"os"
"gitea.maximumdirect.net/eric/scriptorium"
)
func main() {
engine, err := scriptorium.NewEngine(scriptorium.Config{
PromptDir: "./examples/prompts",
ProfileDir: "./examples/profiles",
SchemaDir: "./examples/schemas",
})
if err != nil {
log.Fatal(err)
}
prepared, err := engine.Prepare(context.Background(), scriptorium.RunRequest{
PromptID: "generic.markdown_summary",
Inputs: map[string]scriptorium.ArtifactRef{
"transcript": scriptorium.File("./examples/fixtures/transcript.md"),
"glossary": scriptorium.File("./examples/fixtures/glossary.yml"),
},
})
if err != nil {
log.Fatal(err)
}
summary := struct {
PromptID string `json:"prompt_id"`
SelectedProfileID string `json:"selected_profile_id"`
Model string `json:"model"`
MessageCount int `json:"message_count"`
InputHashes map[string]string `json:"input_hashes"`
}{
PromptID: prepared.PromptID,
SelectedProfileID: prepared.SelectedProfileID,
Model: prepared.EffectiveModelParams.Model,
MessageCount: len(prepared.Messages),
InputHashes: prepared.InputHashes,
}
if err := json.NewEncoder(os.Stdout).Encode(summary); err != nil {
log.Fatal(err)
}
}

51
formatting.go Normal file
View File

@@ -0,0 +1,51 @@
package scriptorium
import "fmt"
// String returns a concise request summary without exposing direct API keys.
func (r RunRequest) String() string {
return r.redactedString()
}
// GoString returns a concise request summary without exposing direct API keys.
func (r RunRequest) GoString() string {
return r.redactedString()
}
func (r RunRequest) redactedString() string {
return fmt.Sprintf(
"scriptorium.RunRequest{PromptID:%q PromptVersion:%q ProfileID:%q APIKeySet:%t Inputs:%d Vars:%d ExecutionSet:%t ValidationSet:%t Metadata:%d}",
r.PromptID,
r.PromptVersion,
r.ProfileID,
r.APIKey != "",
len(r.Inputs),
len(r.Vars),
r.Execution != nil,
r.Validation != nil,
len(r.Metadata),
)
}
// String returns a concise request summary without exposing direct API keys or
// rendered prompt content.
func (r GenerateRequest) String() string {
return r.redactedString()
}
// GoString returns a concise request summary without exposing direct API keys or
// rendered prompt content.
func (r GenerateRequest) GoString() string {
return r.redactedString()
}
func (r GenerateRequest) redactedString() string {
return fmt.Sprintf(
"scriptorium.GenerateRequest{Messages:%d Model:%q APIKeySet:%t StructuredOutputSet:%t ExtraParams:%d}",
len(r.Prompt.Messages),
r.Target.Model,
r.APIKey != "",
r.StructuredOutput != nil,
len(r.Target.ExtraParams),
)
}

View File

@@ -19,7 +19,7 @@ import (
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
renderformat "gitea.maximumdirect.net/eric/scriptorium/internal/format" renderformat "gitea.maximumdirect.net/eric/scriptorium/internal/format"
"gitea.maximumdirect.net/eric/scriptorium/internal/llm" "gitea.maximumdirect.net/eric/scriptorium/internal/llm"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile" "gitea.maximumdirect.net/eric/scriptorium/internal/profile/builtin"
"gitea.maximumdirect.net/eric/scriptorium/internal/prompt" "gitea.maximumdirect.net/eric/scriptorium/internal/prompt"
"gitea.maximumdirect.net/eric/scriptorium/internal/promptdef" "gitea.maximumdirect.net/eric/scriptorium/internal/promptdef"
"gitea.maximumdirect.net/eric/scriptorium/internal/usecase" "gitea.maximumdirect.net/eric/scriptorium/internal/usecase"
@@ -33,8 +33,7 @@ const (
) )
const ( const (
errPromptDirRequired = "prompt directory is required; provide --prompt-dir or config.yml prompt_dir" errPromptDirRequired = "prompt directory is required; provide --prompt-dir or config.yml prompt_dir"
errProfileDirRequired = "profile directory is required; provide --profile-dir or config.yml profile_dir"
) )
type runConfig struct { type runConfig struct {
@@ -75,10 +74,14 @@ type renderConfig struct {
type serveConfig struct { type serveConfig struct {
configPath string configPath string
addr string addr string
promptDir string promptDir string
profileDir string profileDir string
schemaDir string schemaDir string
artifactRoot string
maxRequestBytes int64
maxArtifactBytes int64
maxResponseBytes int64
} }
type commonCommandSettings struct { type commonCommandSettings struct {
@@ -86,6 +89,10 @@ type commonCommandSettings struct {
profileDir string profileDir string
schemaDir string schemaDir string
serverAddr string serverAddr string
artifactRoot string
maxRequestBytes int64
maxArtifactBytes int64
maxResponseBytes int64
defaultRenderFormat renderformat.PreparedRunOutputFormat defaultRenderFormat renderformat.PreparedRunOutputFormat
} }
@@ -203,9 +210,18 @@ func serveCommand(args []string, stderr io.Writer) int {
return ExitRuntimeError return ExitRuntimeError
} }
runner := newRunner(cfg.promptDir, cfg.profileDir, cfg.schemaDir, llmClient) artifactReader, err := artifactadapter.NewRestrictedCompositeReaderWithLimit(cfg.artifactRoot, cfg.maxArtifactBytes)
if err != nil {
fmt.Fprintf(stderr, "artifact root error: %v\n", err)
return ExitRuntimeError
}
h := httpadapter.NewHandler(runner) runner := newRunnerWithArtifactReader(cfg.promptDir, cfg.profileDir, cfg.schemaDir, llmClient, artifactReader)
h := httpadapter.NewHandlerWithOptions(runner, httpadapter.HandlerOptions{
MaxRequestBytes: cfg.maxRequestBytes,
MaxResponseBytes: cfg.maxResponseBytes,
})
srv := &http.Server{ srv := &http.Server{
Addr: cfg.addr, Addr: cfg.addr,
Handler: h, Handler: h,
@@ -282,6 +298,10 @@ func parseServeArgs(args []string) (*serveConfig, error) {
fs.StringVar(&cfg.promptDir, "prompt-dir", "", "directory containing prompt definition YAML files") fs.StringVar(&cfg.promptDir, "prompt-dir", "", "directory containing prompt definition YAML files")
fs.StringVar(&cfg.profileDir, "profile-dir", "", "directory containing execution profile YAML files") fs.StringVar(&cfg.profileDir, "profile-dir", "", "directory containing execution profile YAML files")
fs.StringVar(&cfg.schemaDir, "schema-dir", "", "base directory for validation schemas") fs.StringVar(&cfg.schemaDir, "schema-dir", "", "base directory for validation schemas")
fs.StringVar(&cfg.artifactRoot, "artifact-root", "", "base directory for HTTP file input artifacts")
fs.Int64Var(&cfg.maxRequestBytes, "max-request-bytes", 0, "maximum HTTP request body bytes; 0 disables the limit")
fs.Int64Var(&cfg.maxArtifactBytes, "max-artifact-bytes", 0, "maximum HTTP file artifact bytes; 0 disables the limit")
fs.Int64Var(&cfg.maxResponseBytes, "max-response-bytes", 0, "maximum HTTP response body bytes; 0 disables the limit")
if err := fs.Parse(args); err != nil { if err := fs.Parse(args); err != nil {
return nil, err return nil, err
@@ -291,10 +311,14 @@ func parseServeArgs(args []string) (*serveConfig, error) {
} }
settings, err := resolveCommonSettings(fs, cfg.configPath, appconfig.CLIOverrides{ settings, err := resolveCommonSettings(fs, cfg.configPath, appconfig.CLIOverrides{
PromptDir: cfg.promptDirIfSet(fs), PromptDir: cfg.promptDirIfSet(fs),
ProfileDir: cfg.profileDirIfSet(fs), ProfileDir: cfg.profileDirIfSet(fs),
SchemaDir: cfg.schemaDirIfSet(fs), SchemaDir: cfg.schemaDirIfSet(fs),
ServerAddr: cfg.addrIfSet(fs), ServerAddr: cfg.addrIfSet(fs),
ArtifactRoot: cfg.artifactRootIfSet(fs),
MaxRequestBytes: cfg.maxRequestBytesIfSet(fs),
MaxArtifactBytes: cfg.maxArtifactBytesIfSet(fs),
MaxResponseBytes: cfg.maxResponseBytesIfSet(fs),
}) })
if err != nil { if err != nil {
return nil, err return nil, err
@@ -304,14 +328,23 @@ func parseServeArgs(args []string) (*serveConfig, error) {
cfg.profileDir = settings.profileDir cfg.profileDir = settings.profileDir
cfg.schemaDir = settings.schemaDir cfg.schemaDir = settings.schemaDir
cfg.addr = settings.serverAddr cfg.addr = settings.serverAddr
cfg.artifactRoot = settings.artifactRoot
cfg.maxRequestBytes = settings.maxRequestBytes
cfg.maxArtifactBytes = settings.maxArtifactBytes
cfg.maxResponseBytes = settings.maxResponseBytes
if err := validateRequiredLibraryDirs(cfg.promptDir, cfg.profileDir); err != nil { if err := validateRequiredLibraryDirs(cfg.promptDir); err != nil {
return nil, err return nil, err
} }
cfg.promptDir = filepath.Clean(cfg.promptDir) cfg.promptDir = filepath.Clean(cfg.promptDir)
cfg.profileDir = filepath.Clean(cfg.profileDir) if strings.TrimSpace(cfg.profileDir) != "" {
cfg.profileDir = filepath.Clean(cfg.profileDir)
}
cfg.schemaDir = filepath.Clean(cfg.schemaDir) cfg.schemaDir = filepath.Clean(cfg.schemaDir)
if strings.TrimSpace(cfg.artifactRoot) != "" {
cfg.artifactRoot = filepath.Clean(cfg.artifactRoot)
}
return cfg, nil return cfg, nil
} }
@@ -353,7 +386,7 @@ func finalizeExecutionRequestConfig(fs *flag.FlagSet, cfg *runConfig) error {
cfg.schemaDir = settings.schemaDir cfg.schemaDir = settings.schemaDir
cfg.defaultRenderFormat = settings.defaultRenderFormat cfg.defaultRenderFormat = settings.defaultRenderFormat
if err := validateRequiredLibraryDirs(cfg.promptDir, cfg.profileDir); err != nil { if err := validateRequiredLibraryDirs(cfg.promptDir); err != nil {
return err return err
} }
if strings.TrimSpace(cfg.promptID) == "" { if strings.TrimSpace(cfg.promptID) == "" {
@@ -363,7 +396,9 @@ func finalizeExecutionRequestConfig(fs *flag.FlagSet, cfg *runConfig) error {
return errors.New("at least one --input is required") return errors.New("at least one --input is required")
} }
cfg.promptDir = filepath.Clean(cfg.promptDir) cfg.promptDir = filepath.Clean(cfg.promptDir)
cfg.profileDir = filepath.Clean(cfg.profileDir) if strings.TrimSpace(cfg.profileDir) != "" {
cfg.profileDir = filepath.Clean(cfg.profileDir)
}
if cfg.outputPath != "" { if cfg.outputPath != "" {
cfg.outputPath = filepath.Clean(cfg.outputPath) cfg.outputPath = filepath.Clean(cfg.outputPath)
} }
@@ -426,6 +461,34 @@ func (c *serveConfig) addrIfSet(fs *flag.FlagSet) string {
return "" return ""
} }
func (c *serveConfig) artifactRootIfSet(fs *flag.FlagSet) string {
if flagWasSet(fs, "artifact-root") {
return c.artifactRoot
}
return ""
}
func (c *serveConfig) maxRequestBytesIfSet(fs *flag.FlagSet) *int64 {
if flagWasSet(fs, "max-request-bytes") {
return &c.maxRequestBytes
}
return nil
}
func (c *serveConfig) maxArtifactBytesIfSet(fs *flag.FlagSet) *int64 {
if flagWasSet(fs, "max-artifact-bytes") {
return &c.maxArtifactBytes
}
return nil
}
func (c *serveConfig) maxResponseBytesIfSet(fs *flag.FlagSet) *int64 {
if flagWasSet(fs, "max-response-bytes") {
return &c.maxResponseBytes
}
return nil
}
func registerConfigPathFlag(fs *flag.FlagSet, target *string) { func registerConfigPathFlag(fs *flag.FlagSet, target *string) {
fs.StringVar( fs.StringVar(
target, target,
@@ -463,25 +526,33 @@ func resolveCommonSettings(fs *flag.FlagSet, configPath string, overrides appcon
profileDir: settings.ProfileDir, profileDir: settings.ProfileDir,
schemaDir: settings.SchemaDir, schemaDir: settings.SchemaDir,
serverAddr: settings.ServerAddr, serverAddr: settings.ServerAddr,
artifactRoot: settings.ArtifactRoot,
maxRequestBytes: settings.MaxRequestBytes,
maxArtifactBytes: settings.MaxArtifactBytes,
maxResponseBytes: settings.MaxResponseBytes,
defaultRenderFormat: settings.DefaultRenderFormat, defaultRenderFormat: settings.DefaultRenderFormat,
}, nil }, nil
} }
func validateRequiredLibraryDirs(promptDir, profileDir string) error { func validateRequiredLibraryDirs(promptDir string) error {
if strings.TrimSpace(promptDir) == "" { if strings.TrimSpace(promptDir) == "" {
return errors.New(errPromptDirRequired) return errors.New(errPromptDirRequired)
} }
if strings.TrimSpace(profileDir) == "" {
return errors.New(errProfileDirRequired)
}
return nil return nil
} }
func newRunner(promptDir, profileDir, schemaDir string, llmClient llm.Client) *usecase.Runner { func newRunner(promptDir, profileDir, schemaDir string, llmClient llm.Client) *usecase.Runner {
return newRunnerWithArtifactReader(promptDir, profileDir, schemaDir, llmClient, artifactadapter.NewCompositeReader())
}
func newRunnerWithArtifactReader(promptDir, profileDir, schemaDir string, llmClient llm.Client, artifactReader artifactadapter.Reader) *usecase.Runner {
if artifactReader == nil {
artifactReader = artifactadapter.NewCompositeReader()
}
return usecase.NewRunner( return usecase.NewRunner(
promptdef.NewFilesystemRepository(promptDir), promptdef.NewFilesystemRepository(promptDir),
profile.NewFilesystemRepository(profileDir), builtin.NewRepositoryWithDirectory(profileDir),
artifactadapter.NewCompositeReader(), artifactReader,
prompt.NewGoRenderer(), prompt.NewGoRenderer(),
llmClient, llmClient,
validate.NewStandardValidator(schemaDir), validate.NewStandardValidator(schemaDir),
@@ -513,18 +584,25 @@ func buildRunRequestFromConfig(cfg *runConfig) (domain.RunRequest, error) {
inputs[name] = domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: path} inputs[name] = domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: path}
} }
var modelOverride *domain.ExecutionTarget var modelOverride *domain.ExecutionTargetOverride
if cfg.llmBaseURLSet || cfg.modelSet || cfg.temperatureSet || cfg.maxTokensSet || cfg.topPSet || cfg.apiKeyEnvSet || cfg.timeoutSet { if cfg.llmBaseURLSet || cfg.modelSet || cfg.temperatureSet || cfg.maxTokensSet || cfg.topPSet || cfg.apiKeyEnvSet || cfg.timeoutSet {
modelOverride = &domain.ExecutionTarget{ modelOverride = &domain.ExecutionTargetOverride{
Endpoint: cfg.llmBaseURL, Endpoint: cfg.llmBaseURL,
Model: cfg.model, Model: cfg.model,
Temperature: cfg.temperature, APIKeyEnv: cfg.apiKeyEnv,
MaxTokens: cfg.maxTokens, }
TopP: cfg.topP, if cfg.temperatureSet {
APIKeyEnv: cfg.apiKeyEnv, modelOverride.Temperature = &cfg.temperature
}
if cfg.maxTokensSet {
modelOverride.MaxTokens = &cfg.maxTokens
}
if cfg.topPSet {
modelOverride.TopP = &cfg.topP
} }
if cfg.timeoutSet { if cfg.timeoutSet {
modelOverride.TimeoutSeconds = int(cfg.timeout.Seconds()) timeoutSeconds := int(cfg.timeout.Seconds())
modelOverride.TimeoutSeconds = &timeoutSeconds
} }
} }
@@ -606,7 +684,7 @@ func printSummary(stderr io.Writer, res *domain.RunResult) {
if res == nil { if res == nil {
return return
} }
fmt.Fprintf(stderr, "prompt=%s@%s selected_profile=%s model=%s validation=%s mode=%s validation_errors=%d prompt_hash=%s inputs=%d usage=%d/%d/%d\n", fmt.Fprintf(stderr, "prompt=%s@%s selected_profile=%s model=%s validation=%s mode=%s validation_errors=%d prompt_hash=%s inputs=%d usage=%d/%d/%d",
res.PromptID, res.PromptID,
res.PromptVersion, res.PromptVersion,
res.SelectedProfileID, res.SelectedProfileID,
@@ -620,11 +698,15 @@ func printSummary(stderr io.Writer, res *domain.RunResult) {
res.Usage.CompletionTokens, res.Usage.CompletionTokens,
res.Usage.TotalTokens, res.Usage.TotalTokens,
) )
if res.Usage.CachedTokens != 0 || res.Usage.CacheWriteTokens != 0 {
fmt.Fprintf(stderr, " cached_tokens=%d cache_write_tokens=%d", res.Usage.CachedTokens, res.Usage.CacheWriteTokens)
}
fmt.Fprintln(stderr)
} }
func printUsage(w io.Writer) { func printUsage(w io.Writer) {
fmt.Fprintln(w, "usage: scriptorium <run|render|serve> ...") fmt.Fprintln(w, "usage: scriptorium <run|render|serve> ...")
fmt.Fprintln(w, " run: scriptorium run [--config PATH] [--prompt-dir DIR] [--profile-dir DIR] --prompt ID --input name=path [--input ...] [--profile ID] [--llm-base-url URL] [--model NAME] [--api-key-env ENV] [--temperature N] [--max-tokens N] [--top-p N] [--var k=v] [--out path] [--timeout 10m]") fmt.Fprintln(w, " run: scriptorium run [--config PATH] [--prompt-dir DIR] [--profile-dir DIR] --prompt ID --input name=path [--input ...] [--profile ID] [--llm-base-url URL] [--model NAME] [--api-key-env ENV] [--temperature N] [--max-tokens N] [--top-p N] [--var k=v] [--out path] [--timeout 10m]")
fmt.Fprintln(w, " render: scriptorium render [--config PATH] [--prompt-dir DIR] [--profile-dir DIR] --prompt ID --input name=path [--input ...] [--profile ID] [--llm-base-url URL] [--model NAME] [--api-key-env ENV] [--temperature N] [--max-tokens N] [--top-p N] [--var k=v] [--format text|json] [--out path] [--timeout 10m]") fmt.Fprintln(w, " render: scriptorium render [--config PATH] [--prompt-dir DIR] [--profile-dir DIR] --prompt ID --input name=path [--input ...] [--profile ID] [--llm-base-url URL] [--model NAME] [--api-key-env ENV] [--temperature N] [--max-tokens N] [--top-p N] [--var k=v] [--format text|json] [--out path] [--timeout 10m]")
fmt.Fprintf(w, " serve: scriptorium serve [--config PATH] [--addr %s] [--prompt-dir DIR] [--profile-dir DIR] [--schema-dir DIR]\n", defaults.HTTPAddrDefault) fmt.Fprintf(w, " serve: scriptorium serve [--config PATH] [--addr %s] [--prompt-dir DIR] [--profile-dir DIR] [--schema-dir DIR] [--artifact-root DIR] [--max-request-bytes N] [--max-artifact-bytes N] [--max-response-bytes N]\n", defaults.HTTPAddrDefault)
} }

View File

@@ -74,12 +74,12 @@ func TestParseRunArgsRequiredFlags(t *testing.T) {
t.Fatalf("expected clear prompt-dir guidance, got %v", err) t.Fatalf("expected clear prompt-dir guidance, got %v", err)
} }
_, err = parseRunArgs([]string{"--config", configPath, "--prompt-dir", "./prompts", "--prompt", "p", "--input", "a=b"}) cfg, err := parseRunArgs([]string{"--config", configPath, "--prompt-dir", "./prompts", "--prompt", "p", "--input", "a=b"})
if err == nil { if err != nil {
t.Fatal("expected missing --profile-dir error") t.Fatalf("expected missing --profile-dir to be accepted, got %v", err)
} }
if !strings.Contains(err.Error(), "profile directory is required") { if cfg.profileDir != "" {
t.Fatalf("expected clear profile-dir guidance, got %v", err) t.Fatalf("expected empty profile dir for built-ins, got %q", cfg.profileDir)
} }
_, err = parseRunArgs([]string{"--config", configPath, "--prompt-dir", "./prompts", "--profile-dir", "./profiles", "--input", "a=b"}) _, err = parseRunArgs([]string{"--config", configPath, "--prompt-dir", "./prompts", "--profile-dir", "./profiles", "--input", "a=b"})
@@ -161,17 +161,12 @@ func TestParseServeArgsRequiredFlags(t *testing.T) {
t.Fatalf("expected clear prompt-dir guidance, got %v", err) t.Fatalf("expected clear prompt-dir guidance, got %v", err)
} }
_, err = parseServeArgs([]string{"--config", configPath, "--prompt-dir", "./prompts"}) cfg, err := parseServeArgs([]string{"--config", configPath, "--prompt-dir", "./prompts"})
if err == nil {
t.Fatal("expected missing --profile-dir error")
}
if !strings.Contains(err.Error(), "profile directory is required") {
t.Fatalf("expected clear profile-dir guidance, got %v", err)
}
cfg, err := parseServeArgs([]string{"--config", configPath, "--prompt-dir", "./prompts", "--profile-dir", "./profiles"})
if err != nil { if err != nil {
t.Fatalf("expected valid serve args, got %v", err) t.Fatalf("expected missing --profile-dir to be accepted, got %v", err)
}
if cfg.profileDir != "" {
t.Fatalf("expected empty profile dir for built-ins, got %q", cfg.profileDir)
} }
if cfg.addr != defaults.HTTPAddrDefault { if cfg.addr != defaults.HTTPAddrDefault {
t.Fatalf("expected default addr %s, got %q", defaults.HTTPAddrDefault, cfg.addr) t.Fatalf("expected default addr %s, got %q", defaults.HTTPAddrDefault, cfg.addr)
@@ -197,6 +192,26 @@ func TestParseServeArgsRejectsRuntimeOverrideFlags(t *testing.T) {
} }
} }
func TestUsageIncludesServeFileAndSizeLimitFlags(t *testing.T) {
var stderr bytes.Buffer
code := Run(nil, io.Discard, &stderr)
if code != ExitRuntimeError {
t.Fatalf("expected usage path to return runtime error, got %d", code)
}
usage := stderr.String()
for _, want := range []string{
"--artifact-root",
"--max-request-bytes",
"--max-artifact-bytes",
"--max-response-bytes",
} {
if !strings.Contains(usage, want) {
t.Fatalf("expected usage to include %q, got %q", want, usage)
}
}
}
func TestParseRunArgsTimeout(t *testing.T) { func TestParseRunArgsTimeout(t *testing.T) {
cfg, err := parseRunArgs([]string{ cfg, err := parseRunArgs([]string{
"--prompt-dir", "./prompts", "--prompt-dir", "./prompts",
@@ -440,11 +455,19 @@ profile_dir: ./from-config/profiles
schema_dir: ./from-config/schemas schema_dir: ./from-config/schemas
server: server:
addr: 127.0.0.1:9000 addr: 127.0.0.1:9000
artifact_root: ./from-config/artifacts
max_request_bytes: 1024
max_artifact_bytes: 2048
max_response_bytes: 4096
`) `)
cfg, err := parseServeArgs([]string{ cfg, err := parseServeArgs([]string{
"--config", configPath, "--config", configPath,
"--addr", ":7777", "--addr", ":7777",
"--artifact-root", "./from-cli/artifacts",
"--max-request-bytes", "0",
"--max-artifact-bytes", "8192",
"--max-response-bytes", "16384",
}) })
if err != nil { if err != nil {
t.Fatalf("expected valid args, got %v", err) t.Fatalf("expected valid args, got %v", err)
@@ -462,6 +485,18 @@ server:
if cfg.addr != ":7777" { if cfg.addr != ":7777" {
t.Fatalf("expected CLI addr override, got %q", cfg.addr) t.Fatalf("expected CLI addr override, got %q", cfg.addr)
} }
if cfg.artifactRoot != filepath.Clean("./from-cli/artifacts") {
t.Fatalf("expected CLI artifact root override, got %q", cfg.artifactRoot)
}
if cfg.maxRequestBytes != 0 {
t.Fatalf("expected CLI max request bytes override, got %d", cfg.maxRequestBytes)
}
if cfg.maxArtifactBytes != 8192 {
t.Fatalf("expected CLI max artifact bytes override, got %d", cfg.maxArtifactBytes)
}
if cfg.maxResponseBytes != 16384 {
t.Fatalf("expected CLI max response bytes override, got %d", cfg.maxResponseBytes)
}
} }
func TestParseServeArgsWithConfigProvidesRequiredDirectoriesAndAddr(t *testing.T) { func TestParseServeArgsWithConfigProvidesRequiredDirectoriesAndAddr(t *testing.T) {
@@ -471,6 +506,10 @@ profile_dir: ./from-config/profiles
schema_dir: ./from-config/schemas schema_dir: ./from-config/schemas
server: server:
addr: 127.0.0.1:9000 addr: 127.0.0.1:9000
artifact_root: ./from-config/artifacts
max_request_bytes: 1024
max_artifact_bytes: 2048
max_response_bytes: 4096
`) `)
cfg, err := parseServeArgs([]string{ cfg, err := parseServeArgs([]string{
@@ -492,6 +531,75 @@ server:
if cfg.addr != "127.0.0.1:9000" { if cfg.addr != "127.0.0.1:9000" {
t.Fatalf("expected addr from config, got %q", cfg.addr) t.Fatalf("expected addr from config, got %q", cfg.addr)
} }
if cfg.artifactRoot != filepath.Clean("./from-config/artifacts") {
t.Fatalf("expected artifact root from config, got %q", cfg.artifactRoot)
}
if cfg.maxRequestBytes != 1024 {
t.Fatalf("expected max request bytes from config, got %d", cfg.maxRequestBytes)
}
if cfg.maxArtifactBytes != 2048 {
t.Fatalf("expected max artifact bytes from config, got %d", cfg.maxArtifactBytes)
}
if cfg.maxResponseBytes != 4096 {
t.Fatalf("expected max response bytes from config, got %d", cfg.maxResponseBytes)
}
}
func TestParseServeArgsRejectsNegativeSizeLimits(t *testing.T) {
tests := []struct {
name string
flag string
}{
{name: "request", flag: "--max-request-bytes"},
{name: "artifact", flag: "--max-artifact-bytes"},
{name: "response", flag: "--max-response-bytes"},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
_, err := parseServeArgs([]string{
"--prompt-dir", "./prompts",
tc.flag, "-1",
})
if err == nil {
t.Fatal("expected negative size limit error")
}
})
}
}
func TestRunAndRenderRejectServeSizeLimitFlags(t *testing.T) {
for _, tc := range []struct {
name string
parse func([]string) error
}{
{
name: "run",
parse: func(args []string) error {
_, err := parseRunArgs(args)
return err
},
},
{
name: "render",
parse: func(args []string) error {
_, err := parseRenderArgs(args)
return err
},
},
} {
t.Run(tc.name, func(t *testing.T) {
err := tc.parse([]string{
"--prompt-dir", "./prompts",
"--prompt", "p",
"--input", "a=b",
"--max-request-bytes", "1024",
})
if err == nil {
t.Fatal("expected unsupported flag error")
}
})
}
} }
func TestRunAndRenderBuildEquivalentRuntimeOverrideRequestsForSharedFlags(t *testing.T) { func TestRunAndRenderBuildEquivalentRuntimeOverrideRequestsForSharedFlags(t *testing.T) {
@@ -565,21 +673,21 @@ profile_dir: ./profiles
} }
} }
func TestParseRunArgsFailsClearlyWhenNoEffectiveProfileDir(t *testing.T) { func TestParseRunArgsAcceptsMissingEffectiveProfileDir(t *testing.T) {
configPath := writeAppConfigFile(t, ` configPath := writeAppConfigFile(t, `
prompt_dir: ./prompts prompt_dir: ./prompts
`) `)
_, err := parseRunArgs([]string{ cfg, err := parseRunArgs([]string{
"--config", configPath, "--config", configPath,
"--prompt", "p", "--prompt", "p",
"--input", "a=b", "--input", "a=b",
}) })
if err == nil { if err != nil {
t.Fatal("expected missing profile_dir error") t.Fatalf("expected missing profile_dir to be accepted, got %v", err)
} }
if !strings.Contains(err.Error(), "profile directory is required") || !strings.Contains(err.Error(), "config.yml profile_dir") { if cfg.profileDir != "" {
t.Fatalf("expected clear profile_dir guidance, got %v", err) t.Fatalf("expected empty profile dir for built-ins, got %q", cfg.profileDir)
} }
} }
@@ -601,21 +709,21 @@ profile_dir: ./profiles
} }
} }
func TestParseRenderArgsFailsClearlyWhenNoEffectiveProfileDir(t *testing.T) { func TestParseRenderArgsAcceptsMissingEffectiveProfileDir(t *testing.T) {
configPath := writeAppConfigFile(t, ` configPath := writeAppConfigFile(t, `
prompt_dir: ./prompts prompt_dir: ./prompts
`) `)
_, err := parseRenderArgs([]string{ cfg, err := parseRenderArgs([]string{
"--config", configPath, "--config", configPath,
"--prompt", "p", "--prompt", "p",
"--input", "a=b", "--input", "a=b",
}) })
if err == nil { if err != nil {
t.Fatal("expected missing profile_dir error") t.Fatalf("expected missing profile_dir to be accepted, got %v", err)
} }
if !strings.Contains(err.Error(), "profile directory is required") || !strings.Contains(err.Error(), "config.yml profile_dir") { if cfg.profileDir != "" {
t.Fatalf("expected clear profile_dir guidance, got %v", err) t.Fatalf("expected empty profile dir for built-ins, got %q", cfg.profileDir)
} }
} }
@@ -747,6 +855,35 @@ func TestRenderCommandDefaultFormatTextIncludesPreparedDetailsAndNoSecrets(t *te
} }
} }
func TestRenderCommandExplicitZeroTemperatureReachesEffectiveSettings(t *testing.T) {
lib := newCLITestLibrary(t)
inputPath := lib.writeInputFile(t, "transcript.md", "hello transcript")
writePromptFile(t, lib.promptDir, "prompt.render", "local-default")
profile := `id: local-default
endpoint: http://127.0.0.1:1/v1
model: profile-model
temperature: 0.7
`
if err := os.WriteFile(filepath.Join(lib.profileDir, "local-default.yaml"), []byte(profile), 0o644); err != nil {
t.Fatalf("failed to write profile fixture: %v", err)
}
code, stdout, stderr := runCLICommand(t, renderCommand, []string{
"--prompt-dir", lib.promptDir,
"--profile-dir", lib.profileDir,
"--prompt", "prompt.render",
"--input", "transcript=" + inputPath,
"--temperature", "0",
})
if code != ExitOK {
t.Fatalf("expected ExitOK, got %d stderr=%q", code, stderr)
}
if !strings.Contains(stdout, "\n temperature: 0\n") {
t.Fatalf("expected explicit zero temperature in effective settings, got:\n%s", stdout)
}
}
func TestRenderCommandSucceedsWithPromptAndProfileDirsFromConfig(t *testing.T) { func TestRenderCommandSucceedsWithPromptAndProfileDirsFromConfig(t *testing.T) {
lib := newCLITestLibrary(t) lib := newCLITestLibrary(t)
inputPath := lib.writeInputFile(t, "transcript.md", "hello transcript") inputPath := lib.writeInputFile(t, "transcript.md", "hello transcript")
@@ -900,6 +1037,29 @@ func TestRenderCommandPromptDefaultProfileWorksThroughCLIPath(t *testing.T) {
} }
} }
func TestRenderCommandUsesBuiltInProfileWithoutProfileDir(t *testing.T) {
t.Setenv("OPENROUTER_API_KEY", "test-key")
lib := newCLITestLibrary(t)
inputPath := lib.writeInputFile(t, "transcript.md", "hello")
writePromptFile(t, lib.promptDir, "prompt.builtin", "mistral-small-3")
code, stdout, stderr := runCLICommand(t, renderCommand, []string{
"--prompt-dir", lib.promptDir,
"--prompt", "prompt.builtin",
"--input", "transcript=" + inputPath,
})
if code != ExitOK {
t.Fatalf("expected ExitOK, got %d stderr=%q", code, stderr)
}
if !strings.Contains(stdout, "selected_profile_id: mistral-small-3") {
t.Fatalf("expected built-in selected profile, got %q", stdout)
}
if !strings.Contains(stdout, "model: mistralai/mistral-small-3.2-24b-instruct") {
t.Fatalf("expected built-in model, got %q", stdout)
}
}
func TestRenderCommandExplicitProfileOverridesPromptDefault(t *testing.T) { func TestRenderCommandExplicitProfileOverridesPromptDefault(t *testing.T) {
lib := newCLITestLibrary(t) lib := newCLITestLibrary(t)
inputPath := lib.writeInputFile(t, "transcript.md", "hello") inputPath := lib.writeInputFile(t, "transcript.md", "hello")
@@ -1064,6 +1224,38 @@ func TestWriteOutputAndSummaryUseSeparateWriters(t *testing.T) {
if !strings.Contains(stderr.String(), "prompt=p@1") { if !strings.Contains(stderr.String(), "prompt=p@1") {
t.Fatalf("expected summary on stderr, got %q", stderr.String()) t.Fatalf("expected summary on stderr, got %q", stderr.String())
} }
if strings.Contains(stderr.String(), "cached_tokens=") || strings.Contains(stderr.String(), "cache_write_tokens=") {
t.Fatalf("expected zero cache usage to be omitted from summary, got %q", stderr.String())
}
}
func TestPrintSummaryIncludesCacheUsageWhenPresent(t *testing.T) {
var stderr bytes.Buffer
printSummary(&stderr, &domain.RunResult{
PromptID: "p",
PromptVersion: "1",
SelectedProfileID: "exec",
ModelName: "m",
Validation: domain.ValidationResult{Status: domain.ValidationPassed, Mode: domain.ValidationBasic},
RenderedPromptHash: "h",
InputHashes: map[string]string{"in": "x"},
Usage: domain.TokenUsage{
PromptTokens: 10,
CompletionTokens: 5,
TotalTokens: 15,
CachedTokens: 0,
CacheWriteTokens: 3,
},
})
summary := stderr.String()
if !strings.Contains(summary, "usage=10/5/15") {
t.Fatalf("expected base usage summary, got %q", summary)
}
if !strings.Contains(summary, "cached_tokens=0 cache_write_tokens=3") {
t.Fatalf("expected cache usage in summary, got %q", summary)
}
} }
type cliTestLibrary struct { type cliTestLibrary struct {

View File

@@ -21,16 +21,16 @@ type inputRefDTO struct {
} }
type modelOverrideRequestDTO struct { type modelOverrideRequestDTO struct {
Endpoint string `json:"endpoint,omitempty"` Endpoint string `json:"endpoint,omitempty"`
Model string `json:"model,omitempty"` Model string `json:"model,omitempty"`
Temperature float64 `json:"temperature,omitempty"` Temperature *float64 `json:"temperature,omitempty"`
MaxTokens int `json:"max_tokens,omitempty"` MaxTokens *int `json:"max_tokens,omitempty"`
TopP float64 `json:"top_p,omitempty"` TopP *float64 `json:"top_p,omitempty"`
TimeoutSeconds int `json:"timeout_seconds,omitempty"` TimeoutSeconds *int `json:"timeout_seconds,omitempty"`
ServiceTier string `json:"service_tier,omitempty"` ServiceTier string `json:"service_tier,omitempty"`
ReasoningEffort string `json:"reasoning_effort,omitempty"` ReasoningEffort string `json:"reasoning_effort,omitempty"`
APIKeyEnv string `json:"api_key_env,omitempty"` APIKeyEnv string `json:"api_key_env,omitempty"`
ExtraParams map[string]string `json:"extra_params,omitempty"` ExtraParams map[string]any `json:"extra_params,omitempty"`
} }
type runResponseDTO struct { type runResponseDTO struct {
@@ -70,22 +70,24 @@ type metadataDTO struct {
} }
type modelParamsDTO struct { type modelParamsDTO struct {
Endpoint string `json:"endpoint"` Endpoint string `json:"endpoint"`
Model string `json:"model"` Model string `json:"model"`
Temperature float64 `json:"temperature"` Temperature float64 `json:"temperature"`
MaxTokens int `json:"max_tokens"` MaxTokens int `json:"max_tokens"`
TopP float64 `json:"top_p"` TopP float64 `json:"top_p"`
TimeoutSeconds int `json:"timeout_seconds"` TimeoutSeconds int `json:"timeout_seconds"`
ServiceTier string `json:"service_tier,omitempty"` ServiceTier string `json:"service_tier,omitempty"`
ReasoningEffort string `json:"reasoning_effort,omitempty"` ReasoningEffort string `json:"reasoning_effort,omitempty"`
APIKeyEnv string `json:"api_key_env,omitempty"` APIKeyEnv string `json:"api_key_env,omitempty"`
ExtraParams map[string]string `json:"extra_params,omitempty"` ExtraParams map[string]any `json:"extra_params,omitempty"`
} }
type tokenUsageDTO struct { type tokenUsageDTO struct {
PromptTokens int `json:"prompt_tokens"` PromptTokens int `json:"prompt_tokens"`
CompletionTokens int `json:"completion_tokens"` CompletionTokens int `json:"completion_tokens"`
TotalTokens int `json:"total_tokens"` TotalTokens int `json:"total_tokens"`
CachedTokens int `json:"cached_tokens"`
CacheWriteTokens int `json:"cache_write_tokens"`
} }
type validationDTO struct { type validationDTO struct {

View File

@@ -4,9 +4,12 @@ import (
"context" "context"
"encoding/json" "encoding/json"
"errors" "errors"
"io"
"net/http" "net/http"
"strings" "strings"
"gitea.maximumdirect.net/eric/scriptorium/internal/artifact"
"gitea.maximumdirect.net/eric/scriptorium/internal/defaults"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile" "gitea.maximumdirect.net/eric/scriptorium/internal/profile"
"gitea.maximumdirect.net/eric/scriptorium/internal/promptdef" "gitea.maximumdirect.net/eric/scriptorium/internal/promptdef"
@@ -18,11 +21,24 @@ type Runner interface {
} }
type Handler struct { type Handler struct {
runner Runner runner Runner
options HandlerOptions
}
type HandlerOptions struct {
MaxRequestBytes int64
MaxResponseBytes int64
} }
func NewHandler(runner Runner) *Handler { func NewHandler(runner Runner) *Handler {
return &Handler{runner: runner} return NewHandlerWithOptions(runner, HandlerOptions{
MaxRequestBytes: defaults.HTTPMaxRequestBytesDefault,
MaxResponseBytes: defaults.HTTPMaxResponseBytesDefault,
})
}
func NewHandlerWithOptions(runner Runner, options HandlerOptions) *Handler {
return &Handler{runner: runner, options: options}
} }
func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) { func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
@@ -36,9 +52,26 @@ func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
} }
var req runRequestDTO var req runRequestDTO
dec := json.NewDecoder(r.Body) body := r.Body
if h.options.MaxRequestBytes > 0 {
body = http.MaxBytesReader(w, r.Body, h.options.MaxRequestBytes)
}
dec := json.NewDecoder(body)
dec.DisallowUnknownFields() dec.DisallowUnknownFields()
if err := dec.Decode(&req); err != nil { if err := dec.Decode(&req); err != nil {
if isRequestTooLarge(err) {
writeError(w, http.StatusRequestEntityTooLarge, "request_too_large", "request body is too large")
return
}
writeError(w, http.StatusBadRequest, "invalid_json", "invalid JSON request body")
return
}
var trailing any
if err := dec.Decode(&trailing); err != io.EOF {
if isRequestTooLarge(err) {
writeError(w, http.StatusRequestEntityTooLarge, "request_too_large", "request body is too large")
return
}
writeError(w, http.StatusBadRequest, "invalid_json", "invalid JSON request body") writeError(w, http.StatusBadRequest, "invalid_json", "invalid JSON request body")
return return
} }
@@ -61,9 +94,9 @@ func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
} }
} }
var model *domain.ExecutionTarget var model *domain.ExecutionTargetOverride
if req.Model != nil { if req.Model != nil {
model = executionTargetFromModelOverrideDTO(req.Model) model = executionTargetOverrideFromModelOverrideDTO(req.Model)
} }
res, err := h.runner.Run(r.Context(), domain.RunRequest{ res, err := h.runner.Run(r.Context(), domain.RunRequest{
@@ -105,6 +138,8 @@ func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
PromptTokens: res.Usage.PromptTokens, PromptTokens: res.Usage.PromptTokens,
CompletionTokens: res.Usage.CompletionTokens, CompletionTokens: res.Usage.CompletionTokens,
TotalTokens: res.Usage.TotalTokens, TotalTokens: res.Usage.TotalTokens,
CachedTokens: res.Usage.CachedTokens,
CacheWriteTokens: res.Usage.CacheWriteTokens,
}, },
StartTime: res.StartTime, StartTime: res.StartTime,
EndTime: res.EndTime, EndTime: res.EndTime,
@@ -118,14 +153,14 @@ func (h *Handler) ServeHTTP(w http.ResponseWriter, r *http.Request) {
raw := res.RawOutput raw := res.RawOutput
resp.RawModelOutput = &raw resp.RawModelOutput = &raw
} }
writeJSON(w, http.StatusOK, resp) writeLimitedJSON(w, http.StatusOK, resp, h.options.MaxResponseBytes)
} }
func executionTargetFromModelOverrideDTO(dto *modelOverrideRequestDTO) *domain.ExecutionTarget { func executionTargetOverrideFromModelOverrideDTO(dto *modelOverrideRequestDTO) *domain.ExecutionTargetOverride {
if dto == nil { if dto == nil {
return nil return nil
} }
return &domain.ExecutionTarget{ return &domain.ExecutionTargetOverride{
Endpoint: dto.Endpoint, Endpoint: dto.Endpoint,
Model: dto.Model, Model: dto.Model,
Temperature: dto.Temperature, Temperature: dto.Temperature,
@@ -173,7 +208,7 @@ func mapRunError(err error) (int, string, string) {
return http.StatusNotFound, "profile_not_found", "execution profile not found" return http.StatusNotFound, "profile_not_found", "execution profile not found"
case errors.Is(err, promptdef.ErrInvalidYAML), errors.Is(err, promptdef.ErrInvalidPromptDefinition): case errors.Is(err, promptdef.ErrInvalidYAML), errors.Is(err, promptdef.ErrInvalidPromptDefinition):
return http.StatusBadRequest, "prompt_load_failed", "failed to load prompt definition" return http.StatusBadRequest, "prompt_load_failed", "failed to load prompt definition"
case errors.Is(err, profile.ErrInvalidYAML), errors.Is(err, profile.ErrInvalidProfile): case errors.Is(err, profile.ErrInvalidYAML), errors.Is(err, profile.ErrInvalidProfile), errors.Is(err, profile.ErrRawAPIKeyNotAllowed):
return http.StatusBadRequest, "profile_load_failed", "failed to load execution profile" return http.StatusBadRequest, "profile_load_failed", "failed to load execution profile"
case errors.Is(err, usecase.ErrProfileRequired): case errors.Is(err, usecase.ErrProfileRequired):
return http.StatusBadRequest, "profile_required", "profile_id is required when prompt default_profile is not set" return http.StatusBadRequest, "profile_required", "profile_id is required when prompt default_profile is not set"
@@ -181,8 +216,14 @@ func mapRunError(err error) (int, string, string) {
return http.StatusBadRequest, "api_key_env_missing", "api_key_env is set but the environment variable is missing" return http.StatusBadRequest, "api_key_env_missing", "api_key_env is set but the environment variable is missing"
case errors.Is(err, usecase.ErrInvalidRequest): case errors.Is(err, usecase.ErrInvalidRequest):
return http.StatusBadRequest, "invalid_request", "invalid run request" return http.StatusBadRequest, "invalid_request", "invalid run request"
case errors.Is(err, usecase.ErrProfileLoad): case errors.Is(err, usecase.ErrPromptLoad):
return http.StatusBadRequest, "prompt_load_failed", "failed to load prompt definition" return http.StatusBadRequest, "prompt_load_failed", "failed to load prompt definition"
case errors.Is(err, usecase.ErrProfileLoad):
return http.StatusBadRequest, "profile_load_failed", "failed to load execution profile"
case errors.Is(err, artifact.ErrFileNotAllowed), errors.Is(err, artifact.ErrFileOutsideRoot):
return http.StatusBadRequest, "artifact_not_allowed", "file input artifact is not allowed"
case errors.Is(err, artifact.ErrFileTooLarge):
return http.StatusRequestEntityTooLarge, "artifact_too_large", "file input artifact is too large"
case errors.Is(err, usecase.ErrArtifactLoad): case errors.Is(err, usecase.ErrArtifactLoad):
return http.StatusBadRequest, "artifact_read_failed", "failed to read input artifact" return http.StatusBadRequest, "artifact_read_failed", "failed to read input artifact"
case errors.Is(err, usecase.ErrPromptRender): case errors.Is(err, usecase.ErrPromptRender):
@@ -197,9 +238,23 @@ func mapRunError(err error) (int, string, string) {
} }
func writeJSON(w http.ResponseWriter, status int, v any) { func writeJSON(w http.ResponseWriter, status int, v any) {
writeLimitedJSON(w, status, v, 0)
}
func writeLimitedJSON(w http.ResponseWriter, status int, v any, maxBytes int64) {
data, err := json.Marshal(v)
if err != nil {
writeError(w, http.StatusInternalServerError, "internal_error", "internal server error")
return
}
data = append(data, '\n')
if maxBytes > 0 && int64(len(data)) > maxBytes {
writeError(w, http.StatusRequestEntityTooLarge, "response_too_large", "response body is too large")
return
}
w.Header().Set("Content-Type", "application/json") w.Header().Set("Content-Type", "application/json")
w.WriteHeader(status) w.WriteHeader(status)
_ = json.NewEncoder(w).Encode(v) _, _ = w.Write(data)
} }
func writeError(w http.ResponseWriter, status int, code, message string) { func writeError(w http.ResponseWriter, status int, code, message string) {
@@ -210,3 +265,8 @@ func writeError(w http.ResponseWriter, status int, code, message string) {
}, },
}) })
} }
func isRequestTooLarge(err error) bool {
var maxBytesErr *http.MaxBytesError
return errors.As(err, &maxBytesErr)
}

View File

@@ -7,12 +7,16 @@ import (
"fmt" "fmt"
"net/http" "net/http"
"net/http/httptest" "net/http/httptest"
"os"
"path/filepath"
"reflect" "reflect"
"strings" "strings"
"testing" "testing"
"time" "time"
"gitea.maximumdirect.net/eric/scriptorium/internal/artifact"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
"gitea.maximumdirect.net/eric/scriptorium/internal/llm"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile" "gitea.maximumdirect.net/eric/scriptorium/internal/profile"
"gitea.maximumdirect.net/eric/scriptorium/internal/promptdef" "gitea.maximumdirect.net/eric/scriptorium/internal/promptdef"
"gitea.maximumdirect.net/eric/scriptorium/internal/usecase" "gitea.maximumdirect.net/eric/scriptorium/internal/usecase"
@@ -32,6 +36,40 @@ func (f *fakeRunner) Run(ctx context.Context, req domain.RunRequest) (*domain.Ru
return f.result, nil return f.result, nil
} }
type handlerPromptRepo struct {
def *domain.PromptDefinition
}
func (r handlerPromptRepo) GetPromptDefinition(ctx context.Context, id string, version string) (*domain.PromptDefinition, error) {
return r.def, nil
}
type handlerProfileRepo struct {
profile *domain.ExecutionProfile
}
func (r handlerProfileRepo) GetProfile(ctx context.Context, id string) (*domain.ExecutionProfile, error) {
return r.profile, nil
}
type handlerArtifactReader struct{}
func (handlerArtifactReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domain.Artifact, error) {
return &domain.Artifact{Name: "input", Body: []byte("input"), Hash: "hash"}, nil
}
type handlerRenderer struct{}
func (handlerRenderer) Render(ctx context.Context, definition *domain.PromptDefinition, inputs map[string]*domain.Artifact, vars map[string]string) (*domain.RenderedPrompt, error) {
return &domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}}, nil
}
type handlerLLMClient struct{}
func (handlerLLMClient) Generate(ctx context.Context, req domain.GenerateRequest) (*domain.GenerateResponse, error) {
return &domain.GenerateResponse{Content: "ok"}, nil
}
func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) { func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
start := time.Now().UTC() start := time.Now().UTC()
end := start.Add(2 * time.Second) end := start.Add(2 * time.Second)
@@ -66,11 +104,17 @@ func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
APIKeyEnv: envName, APIKeyEnv: envName,
}, },
InputHashes: map[string]string{"transcript": "h1"}, InputHashes: map[string]string{"transcript": "h1"},
Usage: domain.TokenUsage{PromptTokens: 1, CompletionTokens: 2, TotalTokens: 3}, Usage: domain.TokenUsage{
StartTime: start, PromptTokens: 1,
EndTime: end, CompletionTokens: 2,
Duration: 2 * time.Second, TotalTokens: 3,
RawOutput: "hello", CachedTokens: 4,
CacheWriteTokens: 5,
},
StartTime: start,
EndTime: end,
Duration: 2 * time.Second,
RawOutput: "hello",
}} }}
h := NewHandler(r) h := NewHandler(r)
@@ -111,6 +155,13 @@ func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
if metadata["model_name"] != "m1" || metadata["endpoint"] != "http://llm/v1" { if metadata["model_name"] != "m1" || metadata["endpoint"] != "http://llm/v1" {
t.Fatalf("unexpected model metadata: name=%#v endpoint=%#v", metadata["model_name"], metadata["endpoint"]) t.Fatalf("unexpected model metadata: name=%#v endpoint=%#v", metadata["model_name"], metadata["endpoint"])
} }
usage := metadata["usage"].(map[string]any)
if usage["prompt_tokens"] != float64(1) || usage["completion_tokens"] != float64(2) || usage["total_tokens"] != float64(3) {
t.Fatalf("unexpected base usage metadata: %#v", usage)
}
if usage["cached_tokens"] != float64(4) || usage["cache_write_tokens"] != float64(5) {
t.Fatalf("unexpected cache usage metadata: %#v", usage)
}
modelParams := metadata["model_params"].(map[string]any) modelParams := metadata["model_params"].(map[string]any)
if modelParams["api_key_env"] != envName { if modelParams["api_key_env"] != envName {
t.Fatalf("expected model_params.api_key_env=%q, got %#v", envName, modelParams["api_key_env"]) t.Fatalf("expected model_params.api_key_env=%q, got %#v", envName, modelParams["api_key_env"])
@@ -134,7 +185,7 @@ func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
if r.last.Execution == nil || r.last.Execution.Model != "gpt-x" { if r.last.Execution == nil || r.last.Execution.Model != "gpt-x" {
t.Fatalf("expected model override, got %#v", r.last.Execution) t.Fatalf("expected model override, got %#v", r.last.Execution)
} }
if r.last.Execution.TimeoutSeconds != 120 { if r.last.Execution.TimeoutSeconds == nil || *r.last.Execution.TimeoutSeconds != 120 {
t.Fatalf("expected timeout_seconds override 120, got %#v", r.last.Execution) t.Fatalf("expected timeout_seconds override 120, got %#v", r.last.Execution)
} }
if r.last.Execution.ServiceTier != "flex" { if r.last.Execution.ServiceTier != "flex" {
@@ -142,6 +193,105 @@ func TestHandlerPostRunsSuccessWithExplicitProfileID(t *testing.T) {
} }
} }
func TestHandlerInlineRefsWorkWithoutArtifactRoot(t *testing.T) {
h := newArtifactRootHandler(t, "")
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id":"p",
"inputs":{"x":{"type":"inline","body":"inline body"}}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusOK {
t.Fatalf("expected 200, got %d body=%s", w.Code, w.Body.String())
}
}
func TestHandlerFileRefsWithoutArtifactRootAreRejected(t *testing.T) {
h := newArtifactRootHandler(t, "")
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id":"p",
"inputs":{"x":{"type":"file","uri":"input.txt"}}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusBadRequest, "artifact_not_allowed")
}
func TestHandlerFileRefsUnderArtifactRootWork(t *testing.T) {
root := t.TempDir()
if err := os.WriteFile(filepath.Join(root, "input.txt"), []byte("allowed"), 0o644); err != nil {
t.Fatal(err)
}
h := newArtifactRootHandler(t, root)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id":"p",
"inputs":{"x":{"type":"file","uri":"input.txt"}}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusOK {
t.Fatalf("expected 200, got %d body=%s", w.Code, w.Body.String())
}
}
func TestHandlerFileRefsAboveArtifactLimitAreRejected(t *testing.T) {
root := t.TempDir()
if err := os.WriteFile(filepath.Join(root, "large.txt"), []byte("123456"), 0o644); err != nil {
t.Fatal(err)
}
h := newArtifactRootHandlerWithLimit(t, root, 5)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id":"p",
"inputs":{"x":{"type":"file","uri":"large.txt"}}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusRequestEntityTooLarge, "artifact_too_large")
}
func TestHandlerFileRefsOutsideArtifactRootAreRejected(t *testing.T) {
root := t.TempDir()
outside := t.TempDir()
if err := os.WriteFile(filepath.Join(outside, "secret.txt"), []byte("denied"), 0o644); err != nil {
t.Fatal(err)
}
h := newArtifactRootHandler(t, root)
tests := []struct {
name string
uri string
}{
{name: "relative traversal", uri: filepath.Join("..", filepath.Base(outside), "secret.txt")},
{name: "absolute outside root", uri: filepath.Join(outside, "secret.txt")},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
body := fmt.Sprintf(`{
"prompt_id":"p",
"inputs":{"x":{"type":"file","uri":%q}}
}`, tc.uri)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(body))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusBadRequest, "artifact_not_allowed")
})
}
}
func TestHandlerPostRunsSuccessUsingPromptDefaultProfile(t *testing.T) { func TestHandlerPostRunsSuccessUsingPromptDefaultProfile(t *testing.T) {
r := &fakeRunner{result: &domain.RunResult{ r := &fakeRunner{result: &domain.RunResult{
Artifact: domain.Artifact{Body: []byte("ok")}, Artifact: domain.Artifact{Body: []byte("ok")},
@@ -171,6 +321,10 @@ func TestHandlerPostRunsSuccessUsingPromptDefaultProfile(t *testing.T) {
if metadata["selected_profile_id"] != "prompt-default" { if metadata["selected_profile_id"] != "prompt-default" {
t.Fatalf("expected selected_profile_id from result, got %#v", metadata["selected_profile_id"]) t.Fatalf("expected selected_profile_id from result, got %#v", metadata["selected_profile_id"])
} }
usage := metadata["usage"].(map[string]any)
if usage["cached_tokens"] != float64(0) || usage["cache_write_tokens"] != float64(0) {
t.Fatalf("expected zero cache usage fields to be included, got %#v", usage)
}
} }
func TestHandlerModelOverrideMapsAllSupportedExecutionFields(t *testing.T) { func TestHandlerModelOverrideMapsAllSupportedExecutionFields(t *testing.T) {
@@ -211,20 +365,136 @@ func TestHandlerModelOverrideMapsAllSupportedExecutionFields(t *testing.T) {
got := r.last.Execution got := r.last.Execution
if got.Endpoint != "http://override/v1" || if got.Endpoint != "http://override/v1" ||
got.Model != "override-model" || got.Model != "override-model" ||
got.Temperature != 0.6 ||
got.MaxTokens != 250 ||
got.TopP != 0.85 ||
got.TimeoutSeconds != 33 ||
got.ServiceTier != "flex" || got.ServiceTier != "flex" ||
got.ReasoningEffort != "medium" || got.ReasoningEffort != "medium" ||
got.APIKeyEnv != "SCRIPTORIUM_API_KEY" { got.APIKeyEnv != "SCRIPTORIUM_API_KEY" {
t.Fatalf("unexpected mapped execution target: %+v", got) t.Fatalf("unexpected mapped execution target: %+v", got)
} }
if !reflect.DeepEqual(got.ExtraParams, map[string]string{"provider_option": "on"}) { if got.Temperature == nil || *got.Temperature != 0.6 {
t.Fatalf("unexpected mapped temperature: %#v", got.Temperature)
}
if got.MaxTokens == nil || *got.MaxTokens != 250 {
t.Fatalf("unexpected mapped max_tokens: %#v", got.MaxTokens)
}
if got.TopP == nil || *got.TopP != 0.85 {
t.Fatalf("unexpected mapped top_p: %#v", got.TopP)
}
if got.TimeoutSeconds == nil || *got.TimeoutSeconds != 33 {
t.Fatalf("unexpected mapped timeout_seconds: %#v", got.TimeoutSeconds)
}
if !reflect.DeepEqual(got.ExtraParams, map[string]any{"provider_option": "on"}) {
t.Fatalf("unexpected mapped extra_params: %#v", got.ExtraParams) t.Fatalf("unexpected mapped extra_params: %#v", got.ExtraParams)
} }
} }
func TestHandlerModelOverrideAcceptsJSONCompatibleExtraParams(t *testing.T) {
r := &fakeRunner{result: &domain.RunResult{
Artifact: domain.Artifact{Body: []byte("ok")},
Validation: domain.ValidationResult{Status: domain.ValidationPassed, Mode: domain.ValidationBasic, IsValid: true},
EffectiveModelParams: domain.ExecutionTarget{Endpoint: "http://llm/v1", Model: "m1"},
}}
h := NewHandler(r)
reqBody := `{
"prompt_id": "prompt-1",
"inputs": {"transcript": {"type": "file", "uri": "./t.md"}},
"model": {
"extra_params": {
"string_value": "enabled",
"number_value": 42,
"boolean_value": true,
"object_value": {"nested": "value", "count": 2},
"array_value": ["first", 3, false]
}
}
}`
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(reqBody))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusOK {
t.Fatalf("expected 200, got %d body=%s", w.Code, w.Body.String())
}
if r.last.Execution == nil {
t.Fatal("expected execution override in run request")
}
want := map[string]any{
"string_value": "enabled",
"number_value": float64(42),
"boolean_value": true,
"object_value": map[string]any{"nested": "value", "count": float64(2)},
"array_value": []any{"first", float64(3), false},
}
if !reflect.DeepEqual(r.last.Execution.ExtraParams, want) {
t.Fatalf("unexpected mapped extra_params:\ngot=%#v\nwant=%#v", r.last.Execution.ExtraParams, want)
}
}
func TestHandlerModelOverrideExplicitZeroTemperatureMapsAsPresent(t *testing.T) {
r := &fakeRunner{result: &domain.RunResult{
Artifact: domain.Artifact{Body: []byte("ok")},
Validation: domain.ValidationResult{Status: domain.ValidationPassed, Mode: domain.ValidationBasic, IsValid: true},
EffectiveModelParams: domain.ExecutionTarget{Endpoint: "http://llm/v1", Model: "m1", Temperature: 0},
}}
h := NewHandler(r)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id": "prompt-1",
"inputs": {"transcript": {"type": "file", "uri": "./t.md"}},
"model": {"temperature": 0}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusOK {
t.Fatalf("expected 200, got %d body=%s", w.Code, w.Body.String())
}
if r.last.Execution == nil || r.last.Execution.Temperature == nil {
t.Fatalf("expected temperature override to be present, got %#v", r.last.Execution)
}
if *r.last.Execution.Temperature != 0 {
t.Fatalf("expected zero temperature override, got %v", *r.last.Execution.Temperature)
}
}
func TestHandlerModelOverrideOmittedTemperatureMapsAsAbsent(t *testing.T) {
r := &fakeRunner{result: &domain.RunResult{
Artifact: domain.Artifact{Body: []byte("ok")},
Validation: domain.ValidationResult{Status: domain.ValidationPassed, Mode: domain.ValidationBasic, IsValid: true},
EffectiveModelParams: domain.ExecutionTarget{Endpoint: "http://llm/v1", Model: "m1", Temperature: 0.7},
}}
h := NewHandler(r)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id": "prompt-1",
"inputs": {"transcript": {"type": "file", "uri": "./t.md"}},
"model": {"model": "override-model"}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusOK {
t.Fatalf("expected 200, got %d body=%s", w.Code, w.Body.String())
}
if r.last.Execution == nil {
t.Fatal("expected model override")
}
if r.last.Execution.Temperature != nil {
t.Fatalf("expected omitted temperature to remain absent, got %#v", r.last.Execution.Temperature)
}
var resp map[string]any
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
t.Fatalf("invalid JSON response: %v", err)
}
metadata := resp["metadata"].(map[string]any)
params := metadata["model_params"].(map[string]any)
if params["temperature"] != 0.7 {
t.Fatalf("expected effective profile/default temperature in response, got %#v", params["temperature"])
}
}
func TestHandlerResponseMetadataModelParamsIncludesAllSupportedFields(t *testing.T) { func TestHandlerResponseMetadataModelParamsIncludesAllSupportedFields(t *testing.T) {
r := &fakeRunner{result: &domain.RunResult{ r := &fakeRunner{result: &domain.RunResult{
Artifact: domain.Artifact{ Artifact: domain.Artifact{
@@ -245,8 +515,10 @@ func TestHandlerResponseMetadataModelParamsIncludesAllSupportedFields(t *testing
ServiceTier: "priority", ServiceTier: "priority",
ReasoningEffort: "high", ReasoningEffort: "high",
APIKeyEnv: "SCRIPTORIUM_API_KEY", APIKeyEnv: "SCRIPTORIUM_API_KEY",
ExtraParams: map[string]string{ ExtraParams: map[string]any{
"provider_option": "on", "provider_option": "on",
"number_value": 42,
"object_value": map[string]any{"nested": "value"},
}, },
}, },
}} }}
@@ -301,6 +573,13 @@ func TestHandlerResponseMetadataModelParamsIncludesAllSupportedFields(t *testing
if extraParams["provider_option"] != "on" { if extraParams["provider_option"] != "on" {
t.Fatalf("unexpected extra_params.provider_option: %#v", extraParams["provider_option"]) t.Fatalf("unexpected extra_params.provider_option: %#v", extraParams["provider_option"])
} }
if extraParams["number_value"] != float64(42) {
t.Fatalf("unexpected extra_params.number_value: %#v", extraParams["number_value"])
}
objectValue, ok := extraParams["object_value"].(map[string]any)
if !ok || objectValue["nested"] != "value" {
t.Fatalf("unexpected extra_params.object_value: %#v", extraParams["object_value"])
}
} }
func TestHandlerInvalidJSON(t *testing.T) { func TestHandlerInvalidJSON(t *testing.T) {
@@ -315,6 +594,69 @@ func TestHandlerInvalidJSON(t *testing.T) {
} }
} }
func TestHandlerRejectsTrailingJSON(t *testing.T) {
h := NewHandler(&fakeRunner{})
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{"prompt_id":"p","inputs":{"x":{"type":"file","uri":"a"}}} {}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusBadRequest, "invalid_json")
}
func TestHandlerRequestTooLarge(t *testing.T) {
h := NewHandlerWithOptions(&fakeRunner{}, HandlerOptions{MaxRequestBytes: 12})
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{"prompt_id":"p","inputs":{"x":{"type":"file","uri":"a"}}}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusRequestEntityTooLarge, "request_too_large")
}
func TestHandlerMalformedJSONBelowLimitStillBadRequest(t *testing.T) {
h := NewHandlerWithOptions(&fakeRunner{}, HandlerOptions{MaxRequestBytes: 1024})
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString("{"))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusBadRequest, "invalid_json")
}
func TestHandlerResponseTooLarge(t *testing.T) {
h := NewHandlerWithOptions(&fakeRunner{result: &domain.RunResult{
Artifact: domain.Artifact{Body: []byte(strings.Repeat("x", 128))},
Validation: domain.ValidationResult{Status: domain.ValidationPassed, Mode: domain.ValidationBasic, IsValid: true},
EffectiveModelParams: domain.ExecutionTarget{Endpoint: "http://llm/v1", Model: "m1"},
}}, HandlerOptions{MaxRequestBytes: 1024, MaxResponseBytes: 64})
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{"prompt_id":"p","inputs":{"x":{"type":"file","uri":"a"}}}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusRequestEntityTooLarge, "response_too_large")
}
func TestHandlerRawOutputDoesNotBypassResponseLimit(t *testing.T) {
h := NewHandlerWithOptions(&fakeRunner{result: &domain.RunResult{
Artifact: domain.Artifact{Body: []byte("ok")},
RawOutput: strings.Repeat("raw", 80),
Validation: domain.ValidationResult{Status: domain.ValidationPassed, Mode: domain.ValidationBasic, IsValid: true},
EffectiveModelParams: domain.ExecutionTarget{Endpoint: "http://llm/v1", Model: "m1"},
}}, HandlerOptions{MaxRequestBytes: 1024, MaxResponseBytes: 128})
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id":"p",
"inputs":{"x":{"type":"file","uri":"a"}},
"include_raw_output":true
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
assertHTTPErrorCode(t, w, http.StatusRequestEntityTooLarge, "response_too_large")
}
func TestHandlerMissingPromptID(t *testing.T) { func TestHandlerMissingPromptID(t *testing.T) {
h := NewHandler(&fakeRunner{}) h := NewHandler(&fakeRunner{})
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{"inputs":{"x":{"type":"file","uri":"a"}}}`)) req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{"inputs":{"x":{"type":"file","uri":"a"}}}`))
@@ -335,6 +677,54 @@ func TestHandlerMissingPromptID(t *testing.T) {
} }
} }
func TestHandlerReservedExtraParamsThroughRunnerMapsToInvalidRequest(t *testing.T) {
llmClient, err := llm.NewOpenAICompatibleClient(llm.OpenAICompatibleConfig{})
if err != nil {
t.Fatal(err)
}
runner := usecase.NewRunner(
handlerPromptRepo{def: &domain.PromptDefinition{
ID: "p",
Version: "1",
DefaultProfile: "exec",
Templates: []domain.PromptMessageTemplate{{Role: "user", Content: "hi"}},
OutputFormat: domain.FormatText,
Validation: domain.OutputContract{Format: domain.FormatText, ValidationMode: domain.ValidationNone},
}},
handlerProfileRepo{profile: &domain.ExecutionProfile{
ID: "exec",
Endpoint: "http://example.invalid/v1",
Model: "model",
}},
handlerArtifactReader{},
handlerRenderer{},
llmClient,
nil,
)
h := NewHandler(runner)
req := httptest.NewRequest(http.MethodPost, "/v1/runs", bytes.NewBufferString(`{
"prompt_id":"p",
"inputs":{"x":{"type":"file","uri":"a"}},
"model":{"extra_params":{"model":"collision"}}
}`))
w := httptest.NewRecorder()
h.ServeHTTP(w, req)
if w.Code != http.StatusBadRequest {
t.Fatalf("expected 400, got %d body=%s", w.Code, w.Body.String())
}
var resp map[string]any
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
t.Fatalf("invalid JSON response: %v", err)
}
errBody := resp["error"].(map[string]any)
if errBody["code"] != "invalid_request" {
t.Fatalf("expected invalid_request code, got %#v", errBody["code"])
}
}
func TestHandlerUsecaseErrorMapping(t *testing.T) { func TestHandlerUsecaseErrorMapping(t *testing.T) {
tests := []struct { tests := []struct {
name string name string
@@ -344,11 +734,13 @@ func TestHandlerUsecaseErrorMapping(t *testing.T) {
message string message string
avoidCause string avoidCause string
}{ }{
{name: "prompt not found", err: wrap(usecase.ErrProfileLoad, promptdef.ErrPromptDefinitionNotFound), status: http.StatusNotFound, code: "prompt_not_found", message: "prompt definition not found"}, {name: "prompt not found", err: wrap(usecase.ErrPromptLoad, promptdef.ErrPromptDefinitionNotFound), status: http.StatusNotFound, code: "prompt_not_found", message: "prompt definition not found"},
{name: "prompt load invalid", err: wrap(usecase.ErrProfileLoad, promptdef.ErrInvalidPromptDefinition), status: http.StatusBadRequest, code: "prompt_load_failed", message: "failed to load prompt definition"}, {name: "prompt load invalid", err: wrap(usecase.ErrPromptLoad, promptdef.ErrInvalidPromptDefinition), status: http.StatusBadRequest, code: "prompt_load_failed", message: "failed to load prompt definition"},
{name: "prompt load generic", err: wrap(usecase.ErrPromptLoad, fmt.Errorf("read failed")), status: http.StatusBadRequest, code: "prompt_load_failed", message: "failed to load prompt definition", avoidCause: "read failed"},
{name: "missing profile/default", err: wrap(usecase.ErrInvalidRequest, usecase.ErrProfileRequired), status: http.StatusBadRequest, code: "profile_required", message: "profile_id is required when prompt default_profile is not set"}, {name: "missing profile/default", err: wrap(usecase.ErrInvalidRequest, usecase.ErrProfileRequired), status: http.StatusBadRequest, code: "profile_required", message: "profile_id is required when prompt default_profile is not set"},
{name: "profile not found", err: wrap(usecase.ErrProfileLoad, profile.ErrProfileNotFound), status: http.StatusNotFound, code: "profile_not_found", message: "execution profile not found"}, {name: "profile not found", err: wrap(usecase.ErrProfileLoad, profile.ErrProfileNotFound), status: http.StatusNotFound, code: "profile_not_found", message: "execution profile not found"},
{name: "profile invalid", err: wrap(usecase.ErrProfileLoad, profile.ErrInvalidProfile), status: http.StatusBadRequest, code: "profile_load_failed", message: "failed to load execution profile"}, {name: "profile invalid", err: wrap(usecase.ErrProfileLoad, profile.ErrInvalidProfile), status: http.StatusBadRequest, code: "profile_load_failed", message: "failed to load execution profile"},
{name: "profile load generic", err: wrap(usecase.ErrProfileLoad, fmt.Errorf("read failed")), status: http.StatusBadRequest, code: "profile_load_failed", message: "failed to load execution profile", avoidCause: "read failed"},
{name: "api key env missing", err: wrap(usecase.ErrInvalidRequest, usecase.ErrAPIKeyEnvMissing), status: http.StatusBadRequest, code: "api_key_env_missing", message: "api_key_env is set but the environment variable is missing"}, {name: "api key env missing", err: wrap(usecase.ErrInvalidRequest, usecase.ErrAPIKeyEnvMissing), status: http.StatusBadRequest, code: "api_key_env_missing", message: "api_key_env is set but the environment variable is missing"},
{name: "artifact", err: wrap(usecase.ErrArtifactLoad, fmt.Errorf("read failed")), status: http.StatusBadRequest, code: "artifact_read_failed", message: "failed to read input artifact", avoidCause: "read failed"}, {name: "artifact", err: wrap(usecase.ErrArtifactLoad, fmt.Errorf("read failed")), status: http.StatusBadRequest, code: "artifact_read_failed", message: "failed to read input artifact", avoidCause: "read failed"},
{name: "prompt render", err: wrap(usecase.ErrPromptRender, fmt.Errorf("render failed")), status: http.StatusBadRequest, code: "prompt_render_failed", message: "failed to render prompt", avoidCause: "render failed"}, {name: "prompt render", err: wrap(usecase.ErrPromptRender, fmt.Errorf("render failed")), status: http.StatusBadRequest, code: "prompt_render_failed", message: "failed to render prompt", avoidCause: "render failed"},
@@ -455,3 +847,54 @@ func TestHandlerValidationFailureStillSuccessAndRawOutputOptIn(t *testing.T) {
func wrap(stage error, cause error) error { func wrap(stage error, cause error) error {
return fmt.Errorf("%w: %w", stage, cause) return fmt.Errorf("%w: %w", stage, cause)
} }
func newArtifactRootHandler(t *testing.T, root string) *Handler {
t.Helper()
return newArtifactRootHandlerWithLimit(t, root, 0)
}
func newArtifactRootHandlerWithLimit(t *testing.T, root string, maxArtifactBytes int64) *Handler {
t.Helper()
reader, err := artifact.NewRestrictedCompositeReaderWithLimit(root, maxArtifactBytes)
if err != nil {
t.Fatalf("expected restricted artifact reader: %v", err)
}
runner := usecase.NewRunner(
handlerPromptRepo{def: &domain.PromptDefinition{
ID: "p",
Version: "1",
DefaultProfile: "exec",
Templates: []domain.PromptMessageTemplate{{Role: "user", Content: "hi"}},
OutputFormat: domain.FormatText,
Validation: domain.OutputContract{Format: domain.FormatText, ValidationMode: domain.ValidationNone},
}},
handlerProfileRepo{profile: &domain.ExecutionProfile{
ID: "exec",
Endpoint: "http://example.invalid/v1",
Model: "model",
}},
reader,
handlerRenderer{},
handlerLLMClient{},
nil,
)
return NewHandler(runner)
}
func assertHTTPErrorCode(t *testing.T, w *httptest.ResponseRecorder, status int, code string) {
t.Helper()
if w.Code != status {
t.Fatalf("expected %d, got %d body=%s", status, w.Code, w.Body.String())
}
var resp map[string]any
if err := json.Unmarshal(w.Body.Bytes(), &resp); err != nil {
t.Fatalf("invalid JSON response: %v", err)
}
errBody := resp["error"].(map[string]any)
if errBody["code"] != code {
t.Fatalf("expected code %q, got %#v", code, errBody["code"])
}
}

View File

@@ -7,15 +7,20 @@ import (
"fmt" "fmt"
"gitea.maximumdirect.net/eric/scriptorium/internal/defaults" "gitea.maximumdirect.net/eric/scriptorium/internal/defaults"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
"io"
"mime" "mime"
"os" "os"
"path/filepath" "path/filepath"
"strings"
) )
var ( var (
ErrUnsupportedRefType = errors.New("unsupported artifact reference type") ErrUnsupportedRefType = errors.New("unsupported artifact reference type")
ErrMissingInlineBody = errors.New("missing body for inline artifact") ErrMissingInlineBody = errors.New("missing body for inline artifact")
ErrMissingFilePath = errors.New("missing file path for file artifact") ErrMissingFilePath = errors.New("missing file path for file artifact")
ErrFileNotAllowed = errors.New("file artifact references are not allowed")
ErrFileOutsideRoot = errors.New("file artifact path is outside artifact root")
ErrFileTooLarge = errors.New("file artifact exceeds size limit")
) )
// Reader resolves artifact references into actual artifacts. // Reader resolves artifact references into actual artifacts.
@@ -26,7 +31,7 @@ type Reader interface {
// CompositeReader routes artifact resolution based on the reference type. // CompositeReader routes artifact resolution based on the reference type.
type CompositeReader struct { type CompositeReader struct {
inlineReader *inlineReader inlineReader *inlineReader
fileReader *fileReader fileReader Reader
} }
func NewCompositeReader() Reader { func NewCompositeReader() Reader {
@@ -36,6 +41,21 @@ func NewCompositeReader() Reader {
} }
} }
func NewRestrictedCompositeReader(root string) (Reader, error) {
return NewRestrictedCompositeReaderWithLimit(root, 0)
}
func NewRestrictedCompositeReaderWithLimit(root string, maxBytes int64) (Reader, error) {
fileReader, err := newRestrictedFileReader(root, maxBytes)
if err != nil {
return nil, err
}
return &CompositeReader{
inlineReader: &inlineReader{},
fileReader: fileReader,
}, nil
}
func (c *CompositeReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domain.Artifact, error) { func (c *CompositeReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domain.Artifact, error) {
select { select {
case <-ctx.Done(): case <-ctx.Done():
@@ -89,21 +109,133 @@ func (r *fileReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domain.
return nil, ErrMissingFilePath return nil, ErrMissingFilePath
} }
data, err := os.ReadFile(ref.URI) return readFileArtifact(ref.URI)
if err != nil { }
return nil, fmt.Errorf("failed to read file %s: %w", ref.URI, err)
type deniedFileReader struct{}
func (r deniedFileReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domain.Artifact, error) {
select {
case <-ctx.Done():
return nil, ctx.Err()
default:
} }
contentType := mime.TypeByExtension(filepath.Ext(ref.URI)) if ref.URI == "" {
return nil, ErrMissingFilePath
}
return nil, ErrFileNotAllowed
}
type restrictedFileReader struct {
root string
maxBytes int64
}
func newRestrictedFileReader(root string, maxBytes int64) (Reader, error) {
if maxBytes < 0 {
return nil, fmt.Errorf("artifact size limit must be greater than or equal to 0")
}
cleanRoot := strings.TrimSpace(root)
if cleanRoot == "" {
return deniedFileReader{}, nil
}
absRoot, err := filepath.Abs(filepath.Clean(cleanRoot))
if err != nil {
return nil, fmt.Errorf("resolve artifact root: %w", err)
}
return &restrictedFileReader{root: absRoot, maxBytes: maxBytes}, nil
}
func (r *restrictedFileReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domain.Artifact, error) {
select {
case <-ctx.Done():
return nil, ctx.Err()
default:
}
if ref.URI == "" {
return nil, ErrMissingFilePath
}
path, err := r.resolveLexicalPath(ref.URI)
if err != nil {
return nil, err
}
return readFileArtifactWithLimit(path, r.maxBytes)
}
// resolveLexicalPath checks cleaned path containment without resolving symlinks.
func (r *restrictedFileReader) resolveLexicalPath(rawPath string) (string, error) {
cleanPath := filepath.Clean(strings.TrimSpace(rawPath))
var candidate string
if filepath.IsAbs(cleanPath) {
candidate = cleanPath
} else {
candidate = filepath.Join(r.root, cleanPath)
}
absCandidate, err := filepath.Abs(candidate)
if err != nil {
return "", fmt.Errorf("resolve artifact path: %w", err)
}
absCandidate = filepath.Clean(absCandidate)
rel, err := filepath.Rel(r.root, absCandidate)
if err != nil {
return "", fmt.Errorf("compare artifact path to root: %w", err)
}
if rel == ".." || strings.HasPrefix(rel, ".."+string(filepath.Separator)) || filepath.IsAbs(rel) {
return "", ErrFileOutsideRoot
}
return absCandidate, nil
}
func readFileArtifact(path string) (*domain.Artifact, error) {
return readFileArtifactWithLimit(path, 0)
}
func readFileArtifactWithLimit(path string, maxBytes int64) (*domain.Artifact, error) {
if maxBytes < 0 {
return nil, fmt.Errorf("file size limit must be greater than or equal to 0")
}
file, err := os.Open(path)
if err != nil {
return nil, fmt.Errorf("failed to read file %s: %w", path, err)
}
defer file.Close()
info, err := file.Stat()
if err != nil {
return nil, fmt.Errorf("failed to stat file %s: %w", path, err)
}
if maxBytes > 0 && info.Size() > maxBytes {
return nil, ErrFileTooLarge
}
var reader io.Reader = file
if maxBytes > 0 {
reader = io.LimitReader(file, maxBytes+1)
}
data, err := io.ReadAll(reader)
if err != nil {
return nil, fmt.Errorf("failed to read file %s: %w", path, err)
}
if maxBytes > 0 && int64(len(data)) > maxBytes {
return nil, ErrFileTooLarge
}
contentType := mime.TypeByExtension(filepath.Ext(path))
if contentType == "" { if contentType == "" {
contentType = defaults.ContentTypeTextPlain contentType = defaults.ContentTypeTextPlain
} }
return &domain.Artifact{ return &domain.Artifact{
Name: filepath.Base(ref.URI), Name: filepath.Base(path),
ContentType: contentType, ContentType: contentType,
Body: data, Body: data,
URI: ref.URI, URI: path,
Size: int64(len(data)), Size: int64(len(data)),
Hash: fmt.Sprintf("%x", sha256.Sum256(data)), Hash: fmt.Sprintf("%x", sha256.Sum256(data)),
}, nil }, nil

View File

@@ -4,6 +4,7 @@ import (
"context" "context"
"errors" "errors"
"os" "os"
"path/filepath"
"testing" "testing"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
@@ -56,6 +57,157 @@ func TestCompositeReader_Read(t *testing.T) {
}) })
} }
func TestRestrictedCompositeReader(t *testing.T) {
ctx := context.Background()
root := t.TempDir()
outside := t.TempDir()
if err := os.WriteFile(filepath.Join(root, "input.txt"), []byte("allowed"), 0o644); err != nil {
t.Fatal(err)
}
if err := os.Mkdir(filepath.Join(root, "nested"), 0o755); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(outside, "secret.txt"), []byte("denied"), 0o644); err != nil {
t.Fatal(err)
}
reader, err := NewRestrictedCompositeReader(root)
if err != nil {
t.Fatalf("expected restricted reader construction, got %v", err)
}
t.Run("accepts relative contained path", func(t *testing.T) {
art, err := reader.Read(ctx, domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: "nested/../input.txt"})
if err != nil {
t.Fatalf("expected contained relative path to succeed, got %v", err)
}
if string(art.Body) != "allowed" {
t.Fatalf("unexpected artifact body: %q", string(art.Body))
}
})
t.Run("accepts absolute contained path", func(t *testing.T) {
art, err := reader.Read(ctx, domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: filepath.Join(root, "input.txt")})
if err != nil {
t.Fatalf("expected contained absolute path to succeed, got %v", err)
}
if art.Name != "input.txt" {
t.Fatalf("unexpected artifact name: %q", art.Name)
}
})
t.Run("rejects relative traversal outside root", func(t *testing.T) {
_, err := reader.Read(ctx, domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: filepath.Join("..", filepath.Base(outside), "secret.txt")})
if !errors.Is(err, ErrFileOutsideRoot) {
t.Fatalf("expected ErrFileOutsideRoot, got %v", err)
}
})
t.Run("rejects absolute path outside root", func(t *testing.T) {
_, err := reader.Read(ctx, domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: filepath.Join(outside, "secret.txt")})
if !errors.Is(err, ErrFileOutsideRoot) {
t.Fatalf("expected ErrFileOutsideRoot, got %v", err)
}
})
}
func TestRestrictedCompositeReaderFollowsSymlinkInsideRoot(t *testing.T) {
ctx := context.Background()
root := t.TempDir()
outside := t.TempDir()
target := filepath.Join(outside, "linked.txt")
if err := os.WriteFile(target, []byte("linked outside root"), 0o644); err != nil {
t.Fatal(err)
}
link := filepath.Join(root, "linked.txt")
if err := os.Symlink(target, link); err != nil {
t.Skipf("symlink creation unavailable: %v", err)
}
reader, err := NewRestrictedCompositeReader(root)
if err != nil {
t.Fatalf("expected restricted reader construction, got %v", err)
}
art, err := reader.Read(ctx, domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: "linked.txt"})
if err != nil {
t.Fatalf("expected symlink inside root to be followed, got %v", err)
}
if string(art.Body) != "linked outside root" {
t.Fatalf("unexpected artifact body: %q", string(art.Body))
}
}
func TestRestrictedCompositeReaderWithoutRootDeniesFileRefs(t *testing.T) {
reader, err := NewRestrictedCompositeReader("")
if err != nil {
t.Fatalf("expected restricted reader construction, got %v", err)
}
art, err := reader.Read(context.Background(), domain.ArtifactRef{Type: domain.ArtifactRefInline, Body: "inline"})
if err != nil {
t.Fatalf("expected inline ref to work without artifact root, got %v", err)
}
if string(art.Body) != "inline" {
t.Fatalf("unexpected inline body: %q", string(art.Body))
}
_, err = reader.Read(context.Background(), domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: "input.txt"})
if !errors.Is(err, ErrFileNotAllowed) {
t.Fatalf("expected ErrFileNotAllowed, got %v", err)
}
}
func TestRestrictedCompositeReaderFileSizeLimit(t *testing.T) {
ctx := context.Background()
root := t.TempDir()
if err := os.WriteFile(filepath.Join(root, "exact.txt"), []byte("12345"), 0o644); err != nil {
t.Fatal(err)
}
if err := os.WriteFile(filepath.Join(root, "large.txt"), []byte("123456"), 0o644); err != nil {
t.Fatal(err)
}
reader, err := NewRestrictedCompositeReaderWithLimit(root, 5)
if err != nil {
t.Fatalf("expected restricted reader construction, got %v", err)
}
art, err := reader.Read(ctx, domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: "exact.txt"})
if err != nil {
t.Fatalf("expected file at limit to succeed, got %v", err)
}
if string(art.Body) != "12345" {
t.Fatalf("unexpected artifact body: %q", string(art.Body))
}
_, err = reader.Read(ctx, domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: "large.txt"})
if !errors.Is(err, ErrFileTooLarge) {
t.Fatalf("expected ErrFileTooLarge, got %v", err)
}
}
func TestRestrictedCompositeReaderFileSizeLimitZeroDisablesLimit(t *testing.T) {
root := t.TempDir()
if err := os.WriteFile(filepath.Join(root, "large.txt"), []byte("123456"), 0o644); err != nil {
t.Fatal(err)
}
reader, err := NewRestrictedCompositeReaderWithLimit(root, 0)
if err != nil {
t.Fatalf("expected restricted reader construction, got %v", err)
}
art, err := reader.Read(context.Background(), domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: "large.txt"})
if err != nil {
t.Fatalf("expected unlimited reader to succeed, got %v", err)
}
if string(art.Body) != "123456" {
t.Fatalf("unexpected artifact body: %q", string(art.Body))
}
}
func TestFileReader_Read(t *testing.T) { func TestFileReader_Read(t *testing.T) {
content := []byte("test file content") content := []byte("test file content")
tmpFile, err := os.CreateTemp("", "artifact_test_*.txt") tmpFile, err := os.CreateTemp("", "artifact_test_*.txt")

View File

@@ -40,7 +40,11 @@ type Config struct {
} }
type ServerConfig struct { type ServerConfig struct {
Addr string `yaml:"addr"` Addr string `yaml:"addr"`
ArtifactRoot string `yaml:"artifact_root"`
MaxRequestBytes *int64 `yaml:"max_request_bytes"`
MaxArtifactBytes *int64 `yaml:"max_artifact_bytes"`
MaxResponseBytes *int64 `yaml:"max_response_bytes"`
} }
type DefaultsConfig struct { type DefaultsConfig struct {
@@ -53,16 +57,24 @@ type AppSettings struct {
ProfileDir string ProfileDir string
SchemaDir string SchemaDir string
ServerAddr string ServerAddr string
ArtifactRoot string
MaxRequestBytes int64
MaxArtifactBytes int64
MaxResponseBytes int64
DefaultRenderFormat renderformat.PreparedRunOutputFormat DefaultRenderFormat renderformat.PreparedRunOutputFormat
} }
// CLIOverrides can be applied after config load to enforce precedence. // CLIOverrides can be applied after config load to enforce precedence.
type CLIOverrides struct { type CLIOverrides struct {
PromptDir string PromptDir string
ProfileDir string ProfileDir string
SchemaDir string SchemaDir string
ServerAddr string ServerAddr string
RenderFormat string ArtifactRoot string
MaxRequestBytes *int64
MaxArtifactBytes *int64
MaxResponseBytes *int64
RenderFormat string
} }
// BuiltInDefaults returns compile-time application defaults. // BuiltInDefaults returns compile-time application defaults.
@@ -70,6 +82,9 @@ func BuiltInDefaults() AppSettings {
return AppSettings{ return AppSettings{
SchemaDir: defaults.SchemaDirDefault, SchemaDir: defaults.SchemaDirDefault,
ServerAddr: defaults.HTTPAddrDefault, ServerAddr: defaults.HTTPAddrDefault,
MaxRequestBytes: defaults.HTTPMaxRequestBytesDefault,
MaxArtifactBytes: defaults.HTTPMaxArtifactBytesDefault,
MaxResponseBytes: defaults.HTTPMaxResponseBytesDefault,
DefaultRenderFormat: renderformat.DefaultPreparedRunOutputFormat, DefaultRenderFormat: renderformat.DefaultPreparedRunOutputFormat,
} }
} }
@@ -142,6 +157,27 @@ func ApplyCLIOverrides(base AppSettings, overrides CLIOverrides) (AppSettings, e
if v := strings.TrimSpace(overrides.ServerAddr); v != "" { if v := strings.TrimSpace(overrides.ServerAddr); v != "" {
out.ServerAddr = v out.ServerAddr = v
} }
if v := strings.TrimSpace(overrides.ArtifactRoot); v != "" {
out.ArtifactRoot = filepath.Clean(v)
}
if overrides.MaxRequestBytes != nil {
if *overrides.MaxRequestBytes < 0 {
return AppSettings{}, fmt.Errorf("%w: server.max_request_bytes must be greater than or equal to 0", ErrInvalidConfig)
}
out.MaxRequestBytes = *overrides.MaxRequestBytes
}
if overrides.MaxArtifactBytes != nil {
if *overrides.MaxArtifactBytes < 0 {
return AppSettings{}, fmt.Errorf("%w: server.max_artifact_bytes must be greater than or equal to 0", ErrInvalidConfig)
}
out.MaxArtifactBytes = *overrides.MaxArtifactBytes
}
if overrides.MaxResponseBytes != nil {
if *overrides.MaxResponseBytes < 0 {
return AppSettings{}, fmt.Errorf("%w: server.max_response_bytes must be greater than or equal to 0", ErrInvalidConfig)
}
out.MaxResponseBytes = *overrides.MaxResponseBytes
}
if rawFormat := strings.TrimSpace(overrides.RenderFormat); rawFormat != "" { if rawFormat := strings.TrimSpace(overrides.RenderFormat); rawFormat != "" {
parsed, err := renderformat.ParsePreparedRunOutputFormat(rawFormat) parsed, err := renderformat.ParsePreparedRunOutputFormat(rawFormat)
if err != nil { if err != nil {
@@ -181,6 +217,27 @@ func applyConfig(base AppSettings, cfg Config) (AppSettings, error) {
if v := strings.TrimSpace(cfg.Server.Addr); v != "" { if v := strings.TrimSpace(cfg.Server.Addr); v != "" {
out.ServerAddr = v out.ServerAddr = v
} }
if v := strings.TrimSpace(cfg.Server.ArtifactRoot); v != "" {
out.ArtifactRoot = filepath.Clean(v)
}
if cfg.Server.MaxRequestBytes != nil {
if *cfg.Server.MaxRequestBytes < 0 {
return AppSettings{}, fmt.Errorf("%w: server.max_request_bytes must be greater than or equal to 0", ErrInvalidConfig)
}
out.MaxRequestBytes = *cfg.Server.MaxRequestBytes
}
if cfg.Server.MaxArtifactBytes != nil {
if *cfg.Server.MaxArtifactBytes < 0 {
return AppSettings{}, fmt.Errorf("%w: server.max_artifact_bytes must be greater than or equal to 0", ErrInvalidConfig)
}
out.MaxArtifactBytes = *cfg.Server.MaxArtifactBytes
}
if cfg.Server.MaxResponseBytes != nil {
if *cfg.Server.MaxResponseBytes < 0 {
return AppSettings{}, fmt.Errorf("%w: server.max_response_bytes must be greater than or equal to 0", ErrInvalidConfig)
}
out.MaxResponseBytes = *cfg.Server.MaxResponseBytes
}
if rawFormat := strings.TrimSpace(cfg.Defaults.RenderFormat); rawFormat != "" { if rawFormat := strings.TrimSpace(cfg.Defaults.RenderFormat); rawFormat != "" {
parsed, err := renderformat.ParsePreparedRunOutputFormat(rawFormat) parsed, err := renderformat.ParsePreparedRunOutputFormat(rawFormat)
if err != nil { if err != nil {

View File

@@ -6,6 +6,7 @@ import (
"path/filepath" "path/filepath"
"testing" "testing"
"gitea.maximumdirect.net/eric/scriptorium/internal/defaults"
renderformat "gitea.maximumdirect.net/eric/scriptorium/internal/format" renderformat "gitea.maximumdirect.net/eric/scriptorium/internal/format"
) )
@@ -24,6 +25,20 @@ func TestLoadConfigMissingImplicitPathUsesBuiltInDefaults(t *testing.T) {
} }
} }
func TestBuiltInDefaultsIncludeHTTPSizeLimits(t *testing.T) {
got := BuiltInDefaults()
if got.MaxRequestBytes != defaults.HTTPMaxRequestBytesDefault {
t.Fatalf("unexpected max request bytes: %d", got.MaxRequestBytes)
}
if got.MaxArtifactBytes != defaults.HTTPMaxArtifactBytesDefault {
t.Fatalf("unexpected max artifact bytes: %d", got.MaxArtifactBytes)
}
if got.MaxResponseBytes != defaults.HTTPMaxResponseBytesDefault {
t.Fatalf("unexpected max response bytes: %d", got.MaxResponseBytes)
}
}
func TestLoadConfigMissingExplicitPathReturnsError(t *testing.T) { func TestLoadConfigMissingExplicitPathReturnsError(t *testing.T) {
tmp := t.TempDir() tmp := t.TempDir()
missing := filepath.Join(tmp, "missing.yml") missing := filepath.Join(tmp, "missing.yml")
@@ -92,6 +107,10 @@ profile_dir: ./profiles
schema_dir: ./schemas schema_dir: ./schemas
server: server:
addr: 127.0.0.1:9090 addr: 127.0.0.1:9090
artifact_root: ./artifacts
max_request_bytes: 1024
max_artifact_bytes: 2048
max_response_bytes: 4096
defaults: defaults:
render_format: json render_format: json
`) `)
@@ -113,11 +132,61 @@ defaults:
if got.ServerAddr != "127.0.0.1:9090" { if got.ServerAddr != "127.0.0.1:9090" {
t.Fatalf("unexpected server.addr: %q", got.ServerAddr) t.Fatalf("unexpected server.addr: %q", got.ServerAddr)
} }
if got.ArtifactRoot != filepath.Clean("./artifacts") {
t.Fatalf("unexpected server.artifact_root: %q", got.ArtifactRoot)
}
if got.MaxRequestBytes != 1024 {
t.Fatalf("unexpected server.max_request_bytes: %d", got.MaxRequestBytes)
}
if got.MaxArtifactBytes != 2048 {
t.Fatalf("unexpected server.max_artifact_bytes: %d", got.MaxArtifactBytes)
}
if got.MaxResponseBytes != 4096 {
t.Fatalf("unexpected server.max_response_bytes: %d", got.MaxResponseBytes)
}
if got.DefaultRenderFormat != renderformat.PreparedRunFormatJSON { if got.DefaultRenderFormat != renderformat.PreparedRunFormatJSON {
t.Fatalf("unexpected defaults.render_format: %q", got.DefaultRenderFormat) t.Fatalf("unexpected defaults.render_format: %q", got.DefaultRenderFormat)
} }
} }
func TestLoadConfigAcceptsZeroHTTPSizeLimits(t *testing.T) {
path := writeConfigFile(t, "config.yml", `
server:
max_request_bytes: 0
max_artifact_bytes: 0
max_response_bytes: 0
`)
got, err := LoadConfig(path, true)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if got.MaxRequestBytes != 0 || got.MaxArtifactBytes != 0 || got.MaxResponseBytes != 0 {
t.Fatalf("expected zero limits to be preserved, got request=%d artifact=%d response=%d", got.MaxRequestBytes, got.MaxArtifactBytes, got.MaxResponseBytes)
}
}
func TestLoadConfigRejectsNegativeHTTPSizeLimits(t *testing.T) {
tests := []struct {
name string
body string
}{
{name: "request", body: "server:\n max_request_bytes: -1\n"},
{name: "artifact", body: "server:\n max_artifact_bytes: -1\n"},
{name: "response", body: "server:\n max_response_bytes: -1\n"},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
path := writeConfigFile(t, "config.yml", tc.body)
_, err := LoadConfig(path, true)
if !errors.Is(err, ErrInvalidConfig) {
t.Fatalf("expected ErrInvalidConfig, got %v", err)
}
})
}
}
func TestLoadConfigEmptyFileResolvesToBuiltInDefaults(t *testing.T) { func TestLoadConfigEmptyFileResolvesToBuiltInDefaults(t *testing.T) {
path := writeConfigFile(t, "config.yml", "") path := writeConfigFile(t, "config.yml", "")
@@ -185,15 +254,26 @@ func TestApplyCLIOverridesAppliesPrecedence(t *testing.T) {
ProfileDir: "/from/config/profiles", ProfileDir: "/from/config/profiles",
SchemaDir: "/from/config/schemas", SchemaDir: "/from/config/schemas",
ServerAddr: ":1234", ServerAddr: ":1234",
ArtifactRoot: "/from/config/artifacts",
MaxRequestBytes: 111,
MaxArtifactBytes: 222,
MaxResponseBytes: 333,
DefaultRenderFormat: renderformat.PreparedRunFormatJSON, DefaultRenderFormat: renderformat.PreparedRunFormatJSON,
} }
maxRequestBytes := int64(0)
maxArtifactBytes := int64(444)
maxResponseBytes := int64(555)
got, err := ApplyCLIOverrides(base, CLIOverrides{ got, err := ApplyCLIOverrides(base, CLIOverrides{
PromptDir: "./prompts-cli", PromptDir: "./prompts-cli",
ProfileDir: "./profiles-cli", ProfileDir: "./profiles-cli",
SchemaDir: "./schemas-cli", SchemaDir: "./schemas-cli",
ServerAddr: ":8081", ServerAddr: ":8081",
RenderFormat: "text", ArtifactRoot: "./artifacts-cli",
MaxRequestBytes: &maxRequestBytes,
MaxArtifactBytes: &maxArtifactBytes,
MaxResponseBytes: &maxResponseBytes,
RenderFormat: "text",
}) })
if err != nil { if err != nil {
t.Fatalf("expected no error, got %v", err) t.Fatalf("expected no error, got %v", err)
@@ -211,11 +291,45 @@ func TestApplyCLIOverridesAppliesPrecedence(t *testing.T) {
if got.ServerAddr != ":8081" { if got.ServerAddr != ":8081" {
t.Fatalf("unexpected server addr: %q", got.ServerAddr) t.Fatalf("unexpected server addr: %q", got.ServerAddr)
} }
if got.ArtifactRoot != filepath.Clean("./artifacts-cli") {
t.Fatalf("unexpected artifact root: %q", got.ArtifactRoot)
}
if got.MaxRequestBytes != 0 {
t.Fatalf("unexpected max request bytes: %d", got.MaxRequestBytes)
}
if got.MaxArtifactBytes != 444 {
t.Fatalf("unexpected max artifact bytes: %d", got.MaxArtifactBytes)
}
if got.MaxResponseBytes != 555 {
t.Fatalf("unexpected max response bytes: %d", got.MaxResponseBytes)
}
if got.DefaultRenderFormat != renderformat.PreparedRunFormatText { if got.DefaultRenderFormat != renderformat.PreparedRunFormatText {
t.Fatalf("unexpected render format: %q", got.DefaultRenderFormat) t.Fatalf("unexpected render format: %q", got.DefaultRenderFormat)
} }
} }
func TestApplyCLIOverridesRejectsNegativeHTTPSizeLimits(t *testing.T) {
negative := int64(-1)
tests := []struct {
name string
overrides CLIOverrides
}{
{name: "request", overrides: CLIOverrides{MaxRequestBytes: &negative}},
{name: "artifact", overrides: CLIOverrides{MaxArtifactBytes: &negative}},
{name: "response", overrides: CLIOverrides{MaxResponseBytes: &negative}},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
_, err := ApplyCLIOverrides(BuiltInDefaults(), tc.overrides)
if !errors.Is(err, ErrInvalidConfig) {
t.Fatalf("expected ErrInvalidConfig, got %v", err)
}
})
}
}
func TestApplyCLIOverridesInvalidRenderFormatReturnsError(t *testing.T) { func TestApplyCLIOverridesInvalidRenderFormatReturnsError(t *testing.T) {
_, err := ApplyCLIOverrides(BuiltInDefaults(), CLIOverrides{RenderFormat: "yaml"}) _, err := ApplyCLIOverrides(BuiltInDefaults(), CLIOverrides{RenderFormat: "yaml"})
if err == nil { if err == nil {

View File

@@ -7,13 +7,16 @@ import (
) )
const ( const (
HTTPAddrDefault = ":8080" HTTPAddrDefault = ":8080"
SchemaDirDefault = "." SchemaDirDefault = "."
OutputArtifactName = "output" OutputArtifactName = "output"
ContentTypeTextPlain = "text/plain" ContentTypeTextPlain = "text/plain"
ContentTypeTextMarkdown = "text/markdown" ContentTypeTextMarkdown = "text/markdown"
ContentTypeApplicationJSON = "application/json" ContentTypeApplicationJSON = "application/json"
OpenAIChatCompletionsPath = "/chat/completions" OpenAIChatCompletionsPath = "/chat/completions"
HTTPMaxRequestBytesDefault = 16 * 1024 * 1024
HTTPMaxArtifactBytesDefault = 16 * 1024 * 1024
HTTPMaxResponseBytesDefault = 16 * 1024 * 1024
ExecutionDefaultTemperature = 0.0 ExecutionDefaultTemperature = 0.0
ExecutionDefaultMaxTokens = 0 ExecutionDefaultMaxTokens = 0

View File

@@ -40,14 +40,33 @@ const (
ValidationSkipped ValidationStatus = "skipped" ValidationSkipped ValidationStatus = "skipped"
) )
// CacheControlType defines provider cache behavior for prompt content.
type CacheControlType string
const (
CacheControlEphemeral CacheControlType = "ephemeral"
)
const (
// SessionIDMaxLength is OpenRouter's documented maximum session_id length.
SessionIDMaxLength = 256
)
// CacheControl describes provider cache metadata attached to prompt content.
type CacheControl struct {
Type CacheControlType `yaml:"type" json:"type"`
TTL string `yaml:"ttl,omitempty" json:"ttl,omitempty"`
}
// RunRequest represents a request to generate a single artifact. // RunRequest represents a request to generate a single artifact.
type RunRequest struct { type RunRequest struct {
PromptID string PromptID string
PromptVersion string PromptVersion string
ProfileID string ProfileID string
APIKey string `json:"-" yaml:"-"`
Inputs map[string]ArtifactRef Inputs map[string]ArtifactRef
Vars map[string]string Vars map[string]string
Execution *ExecutionTarget Execution *ExecutionTargetOverride
Validation *OutputContract Validation *OutputContract
Metadata map[string]string Metadata map[string]string
} }
@@ -71,25 +90,26 @@ type RunResult struct {
StartTime time.Time StartTime time.Time
EndTime time.Time EndTime time.Time
Duration time.Duration Duration time.Duration
Error error
} }
// PreparedRun contains pre-LLM execution state from the prepare/render phase. // PreparedRun contains pre-LLM execution state from the prepare/render phase.
// It must never include resolved API key values, model output, or validation data. // It must never include resolved API key values, model output, or validation data.
type PreparedRun struct { type PreparedRun struct {
PromptID string `json:"prompt_id"` PromptID string `json:"prompt_id"`
PromptVersion string `json:"prompt_version,omitempty"` PromptVersion string `json:"prompt_version,omitempty"`
PromptHash string `json:"prompt_hash,omitempty"` PromptHash string `json:"prompt_hash,omitempty"`
SelectedProfileID string `json:"selected_profile_id"` SelectedProfileID string `json:"selected_profile_id"`
EffectiveModelParams ExecutionTarget `json:"effective_model_params"` EffectiveModelParams ExecutionTarget `json:"effective_model_params"`
OutputContract OutputContract `json:"output_contract"` TargetPresence ExecutionTargetPresence `json:"-"`
StructuredOutput *StructuredOutputSpec `json:"structured_output,omitempty"` OutputContract OutputContract `json:"output_contract"`
InputHashes map[string]string `json:"input_hashes,omitempty"` StructuredOutput *StructuredOutputSpec `json:"structured_output,omitempty"`
RenderedPromptHash string `json:"rendered_prompt_hash"` InputHashes map[string]string `json:"input_hashes,omitempty"`
Messages []RenderedMessage `json:"messages"` SessionID string `json:"session_id,omitempty"`
StartTime time.Time `json:"start_time,omitempty"` RenderedPromptHash string `json:"rendered_prompt_hash"`
EndTime time.Time `json:"end_time,omitempty"` Messages []RenderedMessage `json:"messages"`
DurationMS int64 `json:"duration_ms,omitempty"` StartTime time.Time `json:"start_time,omitempty"`
EndTime time.Time `json:"end_time,omitempty"`
DurationMS int64 `json:"duration_ms,omitempty"`
} }
// ArtifactRef represents a reference to an input artifact. // ArtifactRef represents a reference to an input artifact.
@@ -115,6 +135,7 @@ type PromptDefinition struct {
Version string `yaml:"version"` Version string `yaml:"version"`
DefaultProfile string `yaml:"default_profile"` DefaultProfile string `yaml:"default_profile"`
Description string `yaml:"description"` Description string `yaml:"description"`
SessionID string `yaml:"session_id" json:"session_id,omitempty"`
Inputs []PromptInput `yaml:"inputs"` Inputs []PromptInput `yaml:"inputs"`
Templates []PromptMessageTemplate `yaml:"templates"` Templates []PromptMessageTemplate `yaml:"templates"`
OutputFormat OutputFormat `yaml:"output_format"` OutputFormat OutputFormat `yaml:"output_format"`
@@ -131,38 +152,65 @@ type PromptInput struct {
// PromptMessageTemplate defines a template for a chat message. // PromptMessageTemplate defines a template for a chat message.
type PromptMessageTemplate struct { type PromptMessageTemplate struct {
Role string `yaml:"role"` Role string `yaml:"role"`
Content string `yaml:"content"` Content string `yaml:"content"`
ContentFile string `yaml:"content_file"` ContentFile string `yaml:"content_file"`
CacheControl *CacheControl `yaml:"cache_control,omitempty" json:"cache_control,omitempty"`
} }
// ExecutionProfile describes how and where to execute a model. // ExecutionProfile describes how and where to execute a model.
type ExecutionProfile struct { type ExecutionProfile struct {
ID string `yaml:"id"` ID string `yaml:"id"`
Endpoint string `yaml:"endpoint"` Endpoint string `yaml:"endpoint"`
Model string `yaml:"model"` Model string `yaml:"model"`
Temperature float64 `yaml:"temperature"` Temperature float64 `yaml:"temperature"`
MaxTokens int `yaml:"max_tokens"` MaxTokens int `yaml:"max_tokens"`
TopP float64 `yaml:"top_p"` TopP float64 `yaml:"top_p"`
TimeoutSeconds int `yaml:"timeout_seconds"` TimeoutSeconds int `yaml:"timeout_seconds"`
ServiceTier string `yaml:"service_tier"` ServiceTier string `yaml:"service_tier"`
ReasoningEffort string `yaml:"reasoning_effort"` ReasoningEffort string `yaml:"reasoning_effort"`
APIKeyEnv string `yaml:"api_key_env"` APIKeyEnv string `yaml:"api_key_env"`
ExtraParams map[string]string `yaml:"extra_params"` APIKeyRequired bool `yaml:"-" json:"-"`
ExtraParams map[string]any `yaml:"extra_params"`
}
// ExecutionTargetOverride represents per-request runtime setting overrides.
type ExecutionTargetOverride struct {
Endpoint string `json:"endpoint,omitempty"`
Model string `json:"model,omitempty"`
Temperature *float64 `json:"temperature,omitempty"`
MaxTokens *int `json:"max_tokens,omitempty"`
TopP *float64 `json:"top_p,omitempty"`
TimeoutSeconds *int `json:"timeout_seconds,omitempty"`
ServiceTier string `json:"service_tier,omitempty"`
ReasoningEffort string `json:"reasoning_effort,omitempty"`
APIKeyEnv string `json:"api_key_env,omitempty"`
ExtraParams map[string]any `json:"extra_params,omitempty"`
}
// ExecutionTargetPresence tracks which effective runtime fields came from an
// explicit request override even when the resolved value is a zero value.
type ExecutionTargetPresence struct {
Temperature bool
MaxTokens bool
TopP bool
TimeoutSeconds bool
} }
// ExecutionTarget represents effective model runtime settings for a run. // ExecutionTarget represents effective model runtime settings for a run.
type ExecutionTarget struct { type ExecutionTarget struct {
Endpoint string `yaml:"endpoint" json:"endpoint"` Endpoint string `yaml:"endpoint" json:"endpoint"`
Model string `yaml:"model" json:"model"` Model string `yaml:"model" json:"model"`
Temperature float64 `yaml:"temperature" json:"temperature"` Temperature float64 `yaml:"temperature" json:"temperature"`
MaxTokens int `yaml:"max_tokens" json:"max_tokens"` MaxTokens int `yaml:"max_tokens" json:"max_tokens"`
TopP float64 `yaml:"top_p" json:"top_p"` TopP float64 `yaml:"top_p" json:"top_p"`
TimeoutSeconds int `yaml:"timeout_seconds" json:"timeout_seconds"` TimeoutSeconds int `yaml:"timeout_seconds" json:"timeout_seconds"`
ServiceTier string `yaml:"service_tier" json:"service_tier"` ServiceTier string `yaml:"service_tier" json:"service_tier"`
ReasoningEffort string `yaml:"reasoning_effort" json:"reasoning_effort"` ReasoningEffort string `yaml:"reasoning_effort" json:"reasoning_effort"`
APIKeyEnv string `yaml:"api_key_env" json:"api_key_env"` APIKeyEnv string `yaml:"api_key_env" json:"api_key_env"`
ExtraParams map[string]string `yaml:"extra_params" json:"extra_params"` APIKey string `yaml:"-" json:"-"`
APIKeyRequired bool `yaml:"-" json:"-"`
ExtraParams map[string]any `yaml:"extra_params" json:"extra_params"`
} }
// OutputContract defines the requirements for the output artifact. // OutputContract defines the requirements for the output artifact.
@@ -175,19 +223,22 @@ type OutputContract struct {
// RenderedPrompt represents the prompt after template application. // RenderedPrompt represents the prompt after template application.
type RenderedPrompt struct { type RenderedPrompt struct {
Messages []RenderedMessage `json:"messages"` SessionID string `json:"session_id,omitempty"`
Messages []RenderedMessage `json:"messages"`
} }
// RenderedMessage is a single message in a rendered prompt. // RenderedMessage is a single message in a rendered prompt.
type RenderedMessage struct { type RenderedMessage struct {
Role string `json:"role"` Role string `json:"role"`
Content string `json:"content"` Content string `json:"content"`
CacheControl *CacheControl `json:"cache_control,omitempty"`
} }
// GenerateRequest is the internal request passed to the LLM client. // GenerateRequest is the internal request passed to the LLM client.
type GenerateRequest struct { type GenerateRequest struct {
Prompt RenderedPrompt Prompt RenderedPrompt
Target ExecutionTarget Target ExecutionTarget
TargetPresence ExecutionTargetPresence
StructuredOutput *StructuredOutputSpec StructuredOutput *StructuredOutputSpec
} }
@@ -222,6 +273,8 @@ type TokenUsage struct {
PromptTokens int PromptTokens int
CompletionTokens int CompletionTokens int
TotalTokens int TotalTokens int
CachedTokens int
CacheWriteTokens int
} }
// ValidationResult represents the outcome of an output validation. // ValidationResult represents the outcome of an output validation.
@@ -233,23 +286,3 @@ type ValidationResult struct {
RepairAttempts int RepairAttempts int
IsValid bool IsValid bool
} }
// RunMetadata contains auditing information for a run.
type RunMetadata struct {
RunID string
PromptID string
PromptVersion string
PromptHash string
RenderedPromptHash string
SelectedProfileID string
InputHashes map[string]string
ModelEndpoint string
ModelName string
Params ExecutionTarget
Timestamp time.Time
Duration time.Duration
Usage TokenUsage
ValidationMode ValidationMode
ValidationStatus ValidationStatus
RepairAttempts int
}

View File

@@ -20,6 +20,7 @@ func TestPreparedRunJSONDoesNotIncludeSecretValues(t *testing.T) {
Endpoint: "http://llm/v1", Endpoint: "http://llm/v1",
Model: "gpt-test", Model: "gpt-test",
APIKeyEnv: envName, APIKeyEnv: envName,
APIKey: secret,
}, },
InputHashes: map[string]string{"transcript": "hash-1"}, InputHashes: map[string]string{"transcript": "hash-1"},
RenderedPromptHash: "rendered-hash", RenderedPromptHash: "rendered-hash",
@@ -53,3 +54,88 @@ func TestPreparedRunJSONDoesNotIncludeSecretValues(t *testing.T) {
} }
} }
} }
func TestPreparedRunJSONIncludesMessageCacheControlOnlyWhenPresent(t *testing.T) {
prepared := PreparedRun{
PromptID: "prompt.id",
SelectedProfileID: "local-fast",
EffectiveModelParams: ExecutionTarget{
Endpoint: "http://llm/v1",
Model: "gpt-test",
},
RenderedPromptHash: "rendered-hash",
Messages: []RenderedMessage{
{
Role: "system",
Content: "You are helpful.",
CacheControl: &CacheControl{
Type: CacheControlEphemeral,
TTL: "1h",
},
},
{Role: "user", Content: "Summarize this."},
},
}
b, err := json.Marshal(prepared)
if err != nil {
t.Fatalf("marshal failed: %v", err)
}
var decoded struct {
Messages []map[string]any `json:"messages"`
}
if err := json.Unmarshal(b, &decoded); err != nil {
t.Fatalf("unmarshal failed: %v", err)
}
if len(decoded.Messages) != 2 {
t.Fatalf("expected 2 messages, got %d", len(decoded.Messages))
}
cacheControl, ok := decoded.Messages[0]["cache_control"].(map[string]any)
if !ok {
t.Fatalf("expected cache_control on first message, got %#v", decoded.Messages[0])
}
if cacheControl["type"] != string(CacheControlEphemeral) || cacheControl["ttl"] != "1h" {
t.Fatalf("unexpected cache_control payload: %#v", cacheControl)
}
if _, ok := decoded.Messages[1]["cache_control"]; ok {
t.Fatalf("expected second message to omit cache_control, got %#v", decoded.Messages[1])
}
}
func TestPreparedRunJSONIncludesSessionIDOnlyWhenPresent(t *testing.T) {
prepared := PreparedRun{
PromptID: "prompt.id",
SelectedProfileID: "local-fast",
EffectiveModelParams: ExecutionTarget{
Endpoint: "http://llm/v1",
Model: "gpt-test",
},
SessionID: "session-123",
RenderedPromptHash: "rendered-hash",
Messages: []RenderedMessage{{Role: "user", Content: "Summarize this."}},
}
b, err := json.Marshal(prepared)
if err != nil {
t.Fatalf("marshal failed: %v", err)
}
var decoded map[string]any
if err := json.Unmarshal(b, &decoded); err != nil {
t.Fatalf("unmarshal failed: %v", err)
}
if decoded["session_id"] != "session-123" {
t.Fatalf("expected session_id in prepared run JSON, got %#v", decoded["session_id"])
}
prepared.SessionID = ""
b, err = json.Marshal(prepared)
if err != nil {
t.Fatalf("marshal failed: %v", err)
}
if strings.Contains(string(b), "session_id") {
t.Fatalf("expected empty session_id to be omitted, got %s", b)
}
}

View File

@@ -2,7 +2,10 @@ package filecatalog
import ( import (
"context" "context"
"fmt"
"io/fs"
"os" "os"
"path"
"path/filepath" "path/filepath"
"sort" "sort"
"strings" "strings"
@@ -23,7 +26,7 @@ func FindYAMLFiles(ctx context.Context, root string) ([]string, error) {
if d.IsDir() { if d.IsDir() {
return nil return nil
} }
if !isYAMLFile(d.Name()) { if !IsYAMLFile(d.Name()) {
return nil return nil
} }
files = append(files, path) files = append(files, path)
@@ -33,15 +36,100 @@ func FindYAMLFiles(ctx context.Context, root string) ([]string, error) {
return files, err return files, err
} }
// FindFSYAMLFiles returns sorted paths for .yaml and .yml files under root in fsys.
func FindFSYAMLFiles(ctx context.Context, fsys fs.FS, root string) ([]string, error) {
cleanRoot := CleanFSRoot(root)
var files []string
err := fs.WalkDir(fsys, cleanRoot, func(name string, d fs.DirEntry, err error) error {
if err != nil {
return err
}
select {
case <-ctx.Done():
return ctx.Err()
default:
}
if d.IsDir() {
return nil
}
if !IsYAMLFile(d.Name()) {
return nil
}
files = append(files, name)
return nil
})
sort.Strings(files)
return files, err
}
// RelativePath computes a clean relative path from root to path. // RelativePath computes a clean relative path from root to path.
func RelativePath(root string, path string) string { func RelativePath(root string, filePath string) string {
rel, err := filepath.Rel(root, path) rel, err := filepath.Rel(root, filePath)
if err != nil { if err != nil {
return filepath.Clean(path) return filepath.Clean(filePath)
} }
return filepath.Clean(rel) return filepath.Clean(rel)
} }
// CleanFSRoot normalizes a root path for use with fs.FS.
func CleanFSRoot(root string) string {
root = strings.TrimSpace(root)
if root == "" || root == "." {
return "."
}
return path.Clean(root)
}
// DisplayPath returns name relative to root for messages about fs.FS paths.
func DisplayPath(root string, name string) string {
cleanRoot := CleanFSRoot(root)
cleanName := path.Clean(name)
if cleanRoot == "." {
return cleanName
}
prefix := strings.TrimSuffix(cleanRoot, "/") + "/"
if strings.HasPrefix(cleanName, prefix) {
return strings.TrimPrefix(cleanName, prefix)
}
return cleanName
}
// ResolveFSPath resolves userPath from baseDir and keeps it inside root.
func ResolveFSPath(root string, baseDir string, userPath string) (string, string, error) {
cleanRoot := CleanFSRoot(root)
cleanBase := path.Clean(strings.TrimSpace(baseDir))
if cleanBase == "" {
cleanBase = cleanRoot
}
if !containsFSPath(cleanRoot, cleanBase) {
return "", "", fmt.Errorf("base path %q is outside source root %q", cleanBase, cleanRoot)
}
cleanUserPath := strings.TrimSpace(userPath)
if cleanUserPath == "" {
return "", "", fmt.Errorf("path is required")
}
cleanUserPath = path.Clean(cleanUserPath)
if path.IsAbs(cleanUserPath) {
return "", "", fmt.Errorf("path %q must be relative", userPath)
}
resolved := path.Clean(path.Join(cleanBase, cleanUserPath))
if !containsFSPath(cleanRoot, resolved) {
return "", "", fmt.Errorf("path %q escapes source root %q", userPath, cleanRoot)
}
return resolved, DisplayPath(cleanRoot, resolved), nil
}
func containsFSPath(root string, name string) bool {
root = CleanFSRoot(root)
name = path.Clean(name)
if root == "." {
return name == "." || (name != ".." && !strings.HasPrefix(name, "../"))
}
return name == root || strings.HasPrefix(name, strings.TrimSuffix(root, "/")+"/")
}
// Stem strips .yaml or .yml from a file name. // Stem strips .yaml or .yml from a file name.
func Stem(name string) string { func Stem(name string) string {
name = strings.TrimSuffix(name, ".yaml") name = strings.TrimSuffix(name, ".yaml")
@@ -49,6 +137,6 @@ func Stem(name string) string {
return name return name
} }
func isYAMLFile(name string) bool { func IsYAMLFile(name string) bool {
return strings.HasSuffix(name, ".yaml") || strings.HasSuffix(name, ".yml") return strings.HasSuffix(name, ".yaml") || strings.HasSuffix(name, ".yml")
} }

View File

@@ -6,7 +6,9 @@ import (
"os" "os"
"path/filepath" "path/filepath"
"reflect" "reflect"
"strings"
"testing" "testing"
"testing/fstest"
) )
func TestFindYAMLFilesNestedSortedAndFiltered(t *testing.T) { func TestFindYAMLFilesNestedSortedAndFiltered(t *testing.T) {
@@ -43,6 +45,42 @@ func TestFindYAMLFilesHonorsContextCancellation(t *testing.T) {
} }
} }
func TestFindFSYAMLFilesNestedSortedAndFiltered(t *testing.T) {
fsys := fstest.MapFS{
"prompts/z/prompt.yml": &fstest.MapFile{Data: []byte("id: z")},
"prompts/a/profile.yaml": &fstest.MapFile{Data: []byte("id: a")},
"prompts/a/ignore.txt": &fstest.MapFile{Data: []byte("not yaml")},
"prompts/b/ignore.yaml.bak": &fstest.MapFile{Data: []byte("not yaml")},
"other/ignored.yaml": &fstest.MapFile{Data: []byte("id: ignored")},
}
got, err := FindFSYAMLFiles(context.Background(), fsys, " prompts ")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
want := []string{
"prompts/a/profile.yaml",
"prompts/z/prompt.yml",
}
if !reflect.DeepEqual(got, want) {
t.Fatalf("expected sorted YAML files %v, got %v", want, got)
}
}
func TestFindFSYAMLFilesHonorsContextCancellation(t *testing.T) {
fsys := fstest.MapFS{
"one.yaml": &fstest.MapFile{Data: []byte("id: one")},
}
ctx, cancel := context.WithCancel(context.Background())
cancel()
_, err := FindFSYAMLFiles(ctx, fsys, ".")
if !errors.Is(err, context.Canceled) {
t.Fatalf("expected context.Canceled, got %v", err)
}
}
func TestRelativePathNested(t *testing.T) { func TestRelativePathNested(t *testing.T) {
root := t.TempDir() root := t.TempDir()
path := filepath.Join(root, "nested", "profiles", "local.yaml") path := filepath.Join(root, "nested", "profiles", "local.yaml")
@@ -53,6 +91,133 @@ func TestRelativePathNested(t *testing.T) {
} }
} }
func TestCleanFSRoot(t *testing.T) {
tests := []struct {
name string
root string
want string
}{
{name: "empty", root: "", want: "."},
{name: "dot", root: ".", want: "."},
{name: "trimmed", root: " prompts/../profiles ", want: "profiles"},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
if got := CleanFSRoot(tc.root); got != tc.want {
t.Fatalf("expected %q, got %q", tc.want, got)
}
})
}
}
func TestDisplayPath(t *testing.T) {
tests := []struct {
name string
root string
path string
want string
}{
{name: "root dot", root: ".", path: "profiles/local.yaml", want: "profiles/local.yaml"},
{name: "nested root", root: "profiles", path: "profiles/local.yaml", want: "local.yaml"},
{name: "outside root", root: "profiles", path: "other/local.yaml", want: "other/local.yaml"},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
if got := DisplayPath(tc.root, tc.path); got != tc.want {
t.Fatalf("expected %q, got %q", tc.want, got)
}
})
}
}
func TestResolveFSPath(t *testing.T) {
tests := []struct {
name string
root string
baseDir string
userPath string
wantPath string
wantDisplay string
wantErr string
}{
{
name: "sibling inside root",
root: "prompts",
baseDir: "prompts/nested",
userPath: "./messages/user.tmpl",
wantPath: "prompts/nested/messages/user.tmpl",
wantDisplay: "nested/messages/user.tmpl",
},
{
name: "parent inside root",
root: "prompts",
baseDir: "prompts/nested",
userPath: "../shared/user.tmpl",
wantPath: "prompts/shared/user.tmpl",
wantDisplay: "shared/user.tmpl",
},
{
name: "escape rejected",
root: "prompts",
baseDir: "prompts/nested",
userPath: "../../outside.tmpl",
wantErr: "escapes source root",
},
{
name: "absolute path rejected",
root: "prompts",
baseDir: "prompts/nested",
userPath: "/outside.tmpl",
wantErr: "must be relative",
},
{
name: "empty path rejected",
root: "prompts",
baseDir: "prompts/nested",
userPath: " ",
wantErr: "path is required",
},
{
name: "dot root allows normal relative path",
root: ".",
baseDir: ".",
userPath: "schemas/events.schema.json",
wantPath: "schemas/events.schema.json",
wantDisplay: "schemas/events.schema.json",
},
{
name: "dot root rejects parent escape",
root: ".",
baseDir: ".",
userPath: "../outside.tmpl",
wantErr: "escapes source root",
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
gotPath, gotDisplay, err := ResolveFSPath(tc.root, tc.baseDir, tc.userPath)
if tc.wantErr != "" {
if err == nil {
t.Fatalf("expected error containing %q", tc.wantErr)
}
if !strings.Contains(err.Error(), tc.wantErr) {
t.Fatalf("expected error to contain %q, got %v", tc.wantErr, err)
}
return
}
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if gotPath != tc.wantPath || gotDisplay != tc.wantDisplay {
t.Fatalf("expected path/display %q/%q, got %q/%q", tc.wantPath, tc.wantDisplay, gotPath, gotDisplay)
}
})
}
}
func TestStemStripsYAMLExtensions(t *testing.T) { func TestStemStripsYAMLExtensions(t *testing.T) {
tests := []struct { tests := []struct {
name string name string
@@ -73,6 +238,27 @@ func TestStemStripsYAMLExtensions(t *testing.T) {
} }
} }
func TestIsYAMLFile(t *testing.T) {
tests := []struct {
name string
in string
want bool
}{
{name: "yaml", in: "prompt.yaml", want: true},
{name: "yml", in: "profile.yml", want: true},
{name: "backup", in: "profile.yaml.bak", want: false},
{name: "uppercase", in: "profile.YAML", want: false},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
if got := IsYAMLFile(tc.in); got != tc.want {
t.Fatalf("expected %v, got %v", tc.want, got)
}
})
}
}
func mustWriteFile(t *testing.T, path string, content string) { func mustWriteFile(t *testing.T, path string, content string) {
t.Helper() t.Helper()
if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil { if err := os.MkdirAll(filepath.Dir(path), 0o755); err != nil {

View File

@@ -96,6 +96,9 @@ func (textPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, er
if prepared.PromptHash != "" { if prepared.PromptHash != "" {
fmt.Fprintf(&b, "prompt_hash: %s\n", prepared.PromptHash) fmt.Fprintf(&b, "prompt_hash: %s\n", prepared.PromptHash)
} }
if prepared.SessionID != "" {
fmt.Fprintf(&b, "session_id: %s\n", prepared.SessionID)
}
fmt.Fprintf(&b, "rendered_prompt_hash: %s\n", prepared.RenderedPromptHash) fmt.Fprintf(&b, "rendered_prompt_hash: %s\n", prepared.RenderedPromptHash)
target := prepared.EffectiveModelParams target := prepared.EffectiveModelParams
@@ -123,7 +126,11 @@ func (textPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, er
} }
sort.Strings(keys) sort.Strings(keys)
for _, k := range keys { for _, k := range keys {
fmt.Fprintf(&b, " %s: %s\n", k, target.ExtraParams[k]) renderedValue, err := formatExtraParamTextValue(target.ExtraParams[k])
if err != nil {
return nil, fmt.Errorf("failed to format extra_params.%s: %w", k, err)
}
fmt.Fprintf(&b, " %s: %s\n", k, renderedValue)
} }
} }
@@ -151,6 +158,13 @@ func (textPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, er
messages := byRole[role] messages := byRole[role]
for i, msg := range messages { for i, msg := range messages {
fmt.Fprintf(&b, " - message: %d\n", i+1) fmt.Fprintf(&b, " - message: %d\n", i+1)
if msg.CacheControl != nil {
fmt.Fprintf(&b, " cache_control: %s", msg.CacheControl.Type)
if msg.CacheControl.TTL != "" {
fmt.Fprintf(&b, " ttl=%s", msg.CacheControl.TTL)
}
fmt.Fprintln(&b)
}
fmt.Fprintln(&b, " content: |") fmt.Fprintln(&b, " content: |")
content := msg.Content content := msg.Content
if content == "" { if content == "" {
@@ -165,3 +179,15 @@ func (textPreparedRunFormatter) Format(prepared *domain.PreparedRun) ([]byte, er
return b.Bytes(), nil return b.Bytes(), nil
} }
func formatExtraParamTextValue(value any) (string, error) {
if s, ok := value.(string); ok {
return s, nil
}
b, err := json.Marshal(value)
if err != nil {
return "", err
}
return string(b), nil
}

View File

@@ -49,6 +49,36 @@ func TestTextFormatterIncludesPreparedRunDetails(t *testing.T) {
} }
} }
func TestTextFormatterRendersExtraParamsDeterministically(t *testing.T) {
prepared := samplePreparedRun()
prepared.EffectiveModelParams.ExtraParams = map[string]any{
"z_string": "enabled",
"b_number": 42,
"a_object": map[string]any{
"nested": "value",
"count": 2,
},
"c_array": []any{"first", 3, false},
}
out, err := FormatPreparedRun(prepared, PreparedRunFormatText)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
s := string(out)
want := strings.Join([]string{
" extra_params:",
" a_object: {\"count\":2,\"nested\":\"value\"}",
" b_number: 42",
" c_array: [\"first\",3,false]",
" z_string: enabled",
}, "\n")
if !strings.Contains(s, want) {
t.Fatalf("expected deterministic extra_params block %q, got:\n%s", want, s)
}
}
func TestTextFormatterDoesNotIncludeResolvedAPIKeyValue(t *testing.T) { func TestTextFormatterDoesNotIncludeResolvedAPIKeyValue(t *testing.T) {
const secret = "super-secret-api-key" const secret = "super-secret-api-key"
t.Setenv("SCRIPTORIUM_API_KEY", secret) t.Setenv("SCRIPTORIUM_API_KEY", secret)
@@ -62,8 +92,94 @@ func TestTextFormatterDoesNotIncludeResolvedAPIKeyValue(t *testing.T) {
} }
} }
func TestTextFormatterDoesNotIncludeDirectAPIKeyValue(t *testing.T) {
const directKey = "direct-format-key"
prepared := samplePreparedRun()
prepared.EffectiveModelParams.APIKey = directKey
out, err := FormatPreparedRun(prepared, PreparedRunFormatText)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if strings.Contains(string(out), directKey) {
t.Fatalf("text output should not include direct api key value: %s", out)
}
}
func TestTextFormatterIncludesMessageCacheControlBeforeContent(t *testing.T) {
prepared := samplePreparedRun()
prepared.Messages = []domain.RenderedMessage{
{
Role: "system",
Content: "System guidance.",
CacheControl: &domain.CacheControl{
Type: domain.CacheControlEphemeral,
TTL: "1h",
},
},
{Role: "user", Content: "Summarize the transcript."},
}
out, err := FormatPreparedRun(prepared, PreparedRunFormatText)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
s := string(out)
if !strings.Contains(s, " system:\n - message: 1\n cache_control: ephemeral ttl=1h\n content: |") {
t.Fatalf("expected system message cache control before content, got:\n%s", s)
}
if strings.Count(s, "cache_control:") != 1 {
t.Fatalf("expected exactly one cache_control line, got:\n%s", s)
}
}
func TestTextFormatterIncludesSessionIDWhenPresent(t *testing.T) {
prepared := samplePreparedRun()
prepared.SessionID = "session-123"
out, err := FormatPreparedRun(prepared, PreparedRunFormatText)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if !strings.Contains(string(out), "session_id: session-123\n") {
t.Fatalf("expected session_id in text output, got:\n%s", out)
}
}
func TestTextFormatterOmitsEmptyCacheControlTTL(t *testing.T) {
prepared := samplePreparedRun()
prepared.Messages = []domain.RenderedMessage{
{
Role: "system",
Content: "System guidance.",
CacheControl: &domain.CacheControl{
Type: domain.CacheControlEphemeral,
},
},
}
out, err := FormatPreparedRun(prepared, PreparedRunFormatText)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
s := string(out)
if !strings.Contains(s, " cache_control: ephemeral\n") {
t.Fatalf("expected cache_control line without ttl, got:\n%s", s)
}
if strings.Contains(s, "ttl=") {
t.Fatalf("expected empty ttl to be omitted, got:\n%s", s)
}
}
func TestJSONFormatterEmitsValidJSONAndIncludesPreparedRunFields(t *testing.T) { func TestJSONFormatterEmitsValidJSONAndIncludesPreparedRunFields(t *testing.T) {
prepared := samplePreparedRun() prepared := samplePreparedRun()
prepared.SessionID = "session-123"
prepared.EffectiveModelParams.ExtraParams = map[string]any{
"number": 42,
"nested": map[string]any{
"enabled": true,
},
}
out, err := FormatPreparedRun(prepared, PreparedRunFormatJSON) out, err := FormatPreparedRun(prepared, PreparedRunFormatJSON)
if err != nil { if err != nil {
@@ -87,9 +203,24 @@ func TestJSONFormatterEmitsValidJSONAndIncludesPreparedRunFields(t *testing.T) {
if decoded["rendered_prompt_hash"] != "rendered-hash" { if decoded["rendered_prompt_hash"] != "rendered-hash" {
t.Fatalf("expected rendered_prompt_hash in json output, got %#v", decoded["rendered_prompt_hash"]) t.Fatalf("expected rendered_prompt_hash in json output, got %#v", decoded["rendered_prompt_hash"])
} }
if _, ok := decoded["effective_model_params"]; !ok { if decoded["session_id"] != "session-123" {
t.Fatalf("expected session_id in json output, got %#v", decoded["session_id"])
}
modelParams, ok := decoded["effective_model_params"].(map[string]any)
if !ok {
t.Fatalf("expected effective_model_params in json output, got %#v", decoded) t.Fatalf("expected effective_model_params in json output, got %#v", decoded)
} }
extraParams, ok := modelParams["extra_params"].(map[string]any)
if !ok {
t.Fatalf("expected extra_params in json output, got %#v", modelParams["extra_params"])
}
if extraParams["number"] != float64(42) {
t.Fatalf("unexpected numeric extra param in json output: %#v", extraParams["number"])
}
nested, ok := extraParams["nested"].(map[string]any)
if !ok || nested["enabled"] != true {
t.Fatalf("unexpected nested extra param in json output: %#v", extraParams["nested"])
}
if _, ok := decoded["input_hashes"]; !ok { if _, ok := decoded["input_hashes"]; !ok {
t.Fatalf("expected input_hashes in json output, got %#v", decoded) t.Fatalf("expected input_hashes in json output, got %#v", decoded)
} }
@@ -98,6 +229,47 @@ func TestJSONFormatterEmitsValidJSONAndIncludesPreparedRunFields(t *testing.T) {
} }
} }
func TestJSONFormatterIncludesMessageCacheControlOnlyWhenPresent(t *testing.T) {
prepared := samplePreparedRun()
prepared.Messages = []domain.RenderedMessage{
{
Role: "system",
Content: "System guidance.",
CacheControl: &domain.CacheControl{
Type: domain.CacheControlEphemeral,
TTL: "1h",
},
},
{Role: "user", Content: "Summarize the transcript."},
}
out, err := FormatPreparedRun(prepared, PreparedRunFormatJSON)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
var decoded struct {
Messages []map[string]any `json:"messages"`
}
if err := json.Unmarshal(out, &decoded); err != nil {
t.Fatalf("expected valid json output, got %v", err)
}
if len(decoded.Messages) != 2 {
t.Fatalf("expected 2 messages, got %d", len(decoded.Messages))
}
cacheControl, ok := decoded.Messages[0]["cache_control"].(map[string]any)
if !ok {
t.Fatalf("expected first message cache_control, got %#v", decoded.Messages[0])
}
if cacheControl["type"] != string(domain.CacheControlEphemeral) || cacheControl["ttl"] != "1h" {
t.Fatalf("unexpected cache_control payload: %#v", cacheControl)
}
if _, ok := decoded.Messages[1]["cache_control"]; ok {
t.Fatalf("expected second message to omit cache_control, got %#v", decoded.Messages[1])
}
}
func TestJSONFormatterDoesNotIncludeResolvedAPIKeyValue(t *testing.T) { func TestJSONFormatterDoesNotIncludeResolvedAPIKeyValue(t *testing.T) {
const secret = "super-secret-api-key" const secret = "super-secret-api-key"
t.Setenv("SCRIPTORIUM_API_KEY", secret) t.Setenv("SCRIPTORIUM_API_KEY", secret)
@@ -111,6 +283,20 @@ func TestJSONFormatterDoesNotIncludeResolvedAPIKeyValue(t *testing.T) {
} }
} }
func TestJSONFormatterDoesNotIncludeDirectAPIKeyValue(t *testing.T) {
const directKey = "direct-format-key"
prepared := samplePreparedRun()
prepared.EffectiveModelParams.APIKey = directKey
out, err := FormatPreparedRun(prepared, PreparedRunFormatJSON)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if strings.Contains(string(out), directKey) {
t.Fatalf("json output should not include direct api key value: %s", out)
}
}
func TestParsePreparedRunOutputFormatRecognizesSupportedNames(t *testing.T) { func TestParsePreparedRunOutputFormatRecognizesSupportedNames(t *testing.T) {
tests := []struct { tests := []struct {
name string name string

View File

@@ -12,6 +12,7 @@ import (
"os" "os"
"strings" "strings"
"time" "time"
"unicode/utf8"
"gitea.maximumdirect.net/eric/scriptorium/internal/defaults" "gitea.maximumdirect.net/eric/scriptorium/internal/defaults"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
@@ -54,10 +55,11 @@ func NewOpenAICompatibleClient(cfg OpenAICompatibleConfig) (*OpenAICompatibleCli
var client *http.Client var client *http.Client
if cfg.HTTPClient != nil { if cfg.HTTPClient != nil {
client = cfg.HTTPClient cloned := *cfg.HTTPClient
if client.Timeout == 0 { if cloned.Timeout == 0 {
client.Timeout = timeout cloned.Timeout = timeout
} }
client = &cloned
} else { } else {
client = &http.Client{Timeout: timeout} client = &http.Client{Timeout: timeout}
} }
@@ -89,7 +91,12 @@ func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.Genera
return nil, fmt.Errorf("%w: %v", ErrInvalidRequest, err) return nil, fmt.Errorf("%w: %v", ErrInvalidRequest, err)
} }
payload, err := json.Marshal(wireReq) wirePayload, err := openAIChatRequestPayload(wireReq)
if err != nil {
return nil, fmt.Errorf("%w: %v", ErrInvalidRequest, err)
}
payload, err := json.Marshal(wirePayload)
if err != nil { if err != nil {
return nil, fmt.Errorf("%w: failed to encode request: %v", ErrRequestFailed, err) return nil, fmt.Errorf("%w: failed to encode request: %v", ErrRequestFailed, err)
} }
@@ -99,7 +106,9 @@ func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.Genera
return nil, fmt.Errorf("%w: failed to create request: %v", ErrRequestFailed, err) return nil, fmt.Errorf("%w: failed to create request: %v", ErrRequestFailed, err)
} }
httpReq.Header.Set("Content-Type", "application/json") httpReq.Header.Set("Content-Type", "application/json")
if envName := strings.TrimSpace(req.Target.APIKeyEnv); envName != "" { if apiKey := strings.TrimSpace(req.Target.APIKey); apiKey != "" {
httpReq.Header.Set("Authorization", "Bearer "+apiKey)
} else if envName := strings.TrimSpace(req.Target.APIKeyEnv); envName != "" {
apiKey := strings.TrimSpace(os.Getenv(envName)) apiKey := strings.TrimSpace(os.Getenv(envName))
if apiKey == "" { if apiKey == "" {
return nil, fmt.Errorf("%w: api key environment variable %q is not set", ErrInvalidRequest, envName) return nil, fmt.Errorf("%w: api key environment variable %q is not set", ErrInvalidRequest, envName)
@@ -110,6 +119,8 @@ func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.Genera
effectiveTimeout := c.timeout effectiveTimeout := c.timeout
if req.Target.TimeoutSeconds > 0 { if req.Target.TimeoutSeconds > 0 {
effectiveTimeout = time.Duration(req.Target.TimeoutSeconds) * time.Second effectiveTimeout = time.Duration(req.Target.TimeoutSeconds) * time.Second
} else if req.TargetPresence.TimeoutSeconds {
effectiveTimeout = 0
} }
httpClient := c.httpClient httpClient := c.httpClient
@@ -128,8 +139,8 @@ func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.Genera
defer httpResp.Body.Close() defer httpResp.Body.Close()
if httpResp.StatusCode < 200 || httpResp.StatusCode >= 300 { if httpResp.StatusCode < 200 || httpResp.StatusCode >= 300 {
body, _ := io.ReadAll(io.LimitReader(httpResp.Body, 4096)) _, _ = io.Copy(io.Discard, io.LimitReader(httpResp.Body, 4096))
return nil, fmt.Errorf("%w: status=%d body=%q", ErrUnexpectedStatus, httpResp.StatusCode, strings.TrimSpace(string(body))) return nil, fmt.Errorf("%w: status=%d", ErrUnexpectedStatus, httpResp.StatusCode)
} }
var wireResp openAIChatResponse var wireResp openAIChatResponse
@@ -151,6 +162,8 @@ func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.Genera
PromptTokens: wireResp.Usage.PromptTokens, PromptTokens: wireResp.Usage.PromptTokens,
CompletionTokens: wireResp.Usage.CompletionTokens, CompletionTokens: wireResp.Usage.CompletionTokens,
TotalTokens: wireResp.Usage.TotalTokens, TotalTokens: wireResp.Usage.TotalTokens,
CachedTokens: wireResp.Usage.PromptTokensDetails.CachedTokens,
CacheWriteTokens: wireResp.Usage.CacheWriteTokens,
}, },
}, nil }, nil
} }
@@ -167,27 +180,36 @@ func openAIChatRequestFromGenerateRequest(req domain.GenerateRequest, defaultMod
wireReq := openAIChatRequest{ wireReq := openAIChatRequest{
Model: model, Model: model,
} }
if sessionID := strings.TrimSpace(req.Prompt.SessionID); sessionID != "" {
wireReq.Messages = make([]openAIChatMessage, 0, len(req.Prompt.Messages)) if n := utf8.RuneCountInString(sessionID); n > domain.SessionIDMaxLength {
for _, msg := range req.Prompt.Messages { return openAIChatRequest{}, fmt.Errorf("session_id length %d exceeds maximum %d", n, domain.SessionIDMaxLength)
wireReq.Messages = append(wireReq.Messages, openAIChatMessage{ }
Role: msg.Role, wireReq.SessionID = sessionID
Content: msg.Content,
})
} }
if req.Target.Temperature != 0 { wireReq.Messages = make([]openAIChatRequestMessage, 0, len(req.Prompt.Messages))
for _, msg := range req.Prompt.Messages {
wireReq.Messages = append(wireReq.Messages, openAIChatRequestMessageFromRenderedMessage(msg))
}
if req.Target.Temperature != 0 || req.TargetPresence.Temperature {
wireReq.Temperature = &req.Target.Temperature wireReq.Temperature = &req.Target.Temperature
} }
if req.Target.MaxTokens != 0 { if req.Target.MaxTokens != 0 || req.TargetPresence.MaxTokens {
wireReq.MaxTokens = &req.Target.MaxTokens wireReq.MaxTokens = &req.Target.MaxTokens
} }
if req.Target.TopP != 0 { if req.Target.TopP != 0 || req.TargetPresence.TopP {
wireReq.TopP = &req.Target.TopP wireReq.TopP = &req.Target.TopP
} }
if strings.TrimSpace(req.Target.ServiceTier) != "" { if strings.TrimSpace(req.Target.ServiceTier) != "" {
wireReq.ServiceTier = req.Target.ServiceTier wireReq.ServiceTier = req.Target.ServiceTier
} }
if strings.TrimSpace(req.Target.ReasoningEffort) != "" {
wireReq.ReasoningEffort = req.Target.ReasoningEffort
}
if len(req.Target.ExtraParams) > 0 {
wireReq.ExtraParams = req.Target.ExtraParams
}
if req.StructuredOutput != nil { if req.StructuredOutput != nil {
responseFormat, err := toOpenAIResponseFormat(req.StructuredOutput) responseFormat, err := toOpenAIResponseFormat(req.StructuredOutput)
if err != nil { if err != nil {
@@ -200,28 +222,106 @@ func openAIChatRequestFromGenerateRequest(req domain.GenerateRequest, defaultMod
} }
type openAIChatRequest struct { type openAIChatRequest struct {
Model string `json:"model"` Model string `json:"model"`
Messages []openAIChatMessage `json:"messages"` SessionID string `json:"session_id,omitempty"`
Temperature *float64 `json:"temperature,omitempty"` Messages []openAIChatRequestMessage `json:"messages"`
MaxTokens *int `json:"max_tokens,omitempty"` Temperature *float64 `json:"temperature,omitempty"`
TopP *float64 `json:"top_p,omitempty"` MaxTokens *int `json:"max_tokens,omitempty"`
ServiceTier string `json:"service_tier,omitempty"` TopP *float64 `json:"top_p,omitempty"`
ResponseFormat *openAIResponseFormat `json:"response_format,omitempty"` ServiceTier string `json:"service_tier,omitempty"`
ReasoningEffort string `json:"reasoning_effort,omitempty"`
ResponseFormat *openAIResponseFormat `json:"response_format,omitempty"`
ExtraParams map[string]any `json:"-"`
} }
type openAIChatMessage struct { func openAIChatRequestPayload(req openAIChatRequest) (map[string]any, error) {
out := map[string]any{
"model": req.Model,
"messages": req.Messages,
}
if req.SessionID != "" {
out["session_id"] = req.SessionID
}
if req.Temperature != nil {
out["temperature"] = *req.Temperature
}
if req.MaxTokens != nil {
out["max_tokens"] = *req.MaxTokens
}
if req.TopP != nil {
out["top_p"] = *req.TopP
}
if req.ServiceTier != "" {
out["service_tier"] = req.ServiceTier
}
if req.ReasoningEffort != "" {
out["reasoning_effort"] = req.ReasoningEffort
}
if req.ResponseFormat != nil {
out["response_format"] = req.ResponseFormat
}
for key, value := range req.ExtraParams {
if key == "" {
return nil, errors.New("extra_params key must not be empty")
}
if _, reserved := reservedOpenAIChatRequestFields[key]; reserved {
return nil, fmt.Errorf("extra_params key %q collides with reserved request field", key)
}
if _, err := json.Marshal(value); err != nil {
return nil, fmt.Errorf("extra_params.%s must be JSON-serializable: %w", key, err)
}
out[key] = value
}
return out, nil
}
var reservedOpenAIChatRequestFields = map[string]struct{}{
"model": {},
"session_id": {},
"messages": {},
"temperature": {},
"max_tokens": {},
"top_p": {},
"service_tier": {},
"reasoning_effort": {},
"response_format": {},
}
type openAIChatRequestMessage struct {
Role string `json:"role"`
Content any `json:"content"`
}
type openAIChatTextContentBlock struct {
Type string `json:"type"`
Text string `json:"text"`
CacheControl *openAICacheControl `json:"cache_control,omitempty"`
}
type openAICacheControl struct {
Type string `json:"type"`
TTL string `json:"ttl,omitempty"`
}
type openAIChatResponseMessage struct {
Role string `json:"role"` Role string `json:"role"`
Content string `json:"content"` Content string `json:"content"`
} }
type openAIChatResponse struct { type openAIChatResponse struct {
Choices []struct { Choices []struct {
Message openAIChatMessage `json:"message"` Message openAIChatResponseMessage `json:"message"`
} `json:"choices"` } `json:"choices"`
Usage struct { Usage struct {
PromptTokens int `json:"prompt_tokens"` PromptTokens int `json:"prompt_tokens"`
CompletionTokens int `json:"completion_tokens"` CompletionTokens int `json:"completion_tokens"`
TotalTokens int `json:"total_tokens"` TotalTokens int `json:"total_tokens"`
PromptTokensDetails struct {
CachedTokens int `json:"cached_tokens"`
} `json:"prompt_tokens_details"`
CacheWriteTokens int `json:"cache_write_tokens"`
} `json:"usage"` } `json:"usage"`
} }
@@ -236,6 +336,28 @@ type openAIJSONSchemaEnvelope struct {
Schema any `json:"schema"` Schema any `json:"schema"`
} }
func openAIChatRequestMessageFromRenderedMessage(msg domain.RenderedMessage) openAIChatRequestMessage {
wireMsg := openAIChatRequestMessage{
Role: msg.Role,
Content: msg.Content,
}
if msg.CacheControl == nil {
return wireMsg
}
wireMsg.Content = []openAIChatTextContentBlock{
{
Type: "text",
Text: msg.Content,
CacheControl: &openAICacheControl{
Type: string(msg.CacheControl.Type),
TTL: msg.CacheControl.TTL,
},
},
}
return wireMsg
}
func toOpenAIResponseFormat(spec *domain.StructuredOutputSpec) (*openAIResponseFormat, error) { func toOpenAIResponseFormat(spec *domain.StructuredOutputSpec) (*openAIResponseFormat, error) {
if spec == nil { if spec == nil {
return nil, nil return nil, nil

View File

@@ -4,6 +4,7 @@ import (
"context" "context"
"encoding/json" "encoding/json"
"errors" "errors"
"math"
"net/http" "net/http"
"net/http/httptest" "net/http/httptest"
"strings" "strings"
@@ -13,6 +14,64 @@ import (
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
) )
func TestNewOpenAICompatibleClientDoesNotMutateSuppliedZeroTimeoutClient(t *testing.T) {
transport := http.DefaultTransport
supplied := &http.Client{Transport: transport}
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{
HTTPClient: supplied,
})
if err != nil {
t.Fatalf("unexpected constructor error: %v", err)
}
if supplied.Timeout != 0 {
t.Fatalf("expected supplied client timeout to remain zero, got %v", supplied.Timeout)
}
if client.httpClient == supplied {
t.Fatal("expected constructed client to use a cloned HTTP client")
}
if client.httpClient.Timeout != client.timeout {
t.Fatalf("expected cloned client timeout %v, got %v", client.timeout, client.httpClient.Timeout)
}
if client.httpClient.Timeout <= 0 {
t.Fatalf("expected constructed client to use a positive default timeout, got %v", client.httpClient.Timeout)
}
if client.httpClient.Transport != transport {
t.Fatal("expected cloned client to preserve the supplied transport")
}
}
func TestNewOpenAICompatibleClientDoesNotMutateSuppliedNonzeroTimeoutClient(t *testing.T) {
transport := http.DefaultTransport
suppliedTimeout := 37 * time.Second
supplied := &http.Client{
Timeout: suppliedTimeout,
Transport: transport,
}
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{
Timeout: 2 * time.Second,
HTTPClient: supplied,
})
if err != nil {
t.Fatalf("unexpected constructor error: %v", err)
}
if supplied.Timeout != suppliedTimeout {
t.Fatalf("expected supplied client timeout to remain %v, got %v", suppliedTimeout, supplied.Timeout)
}
if client.httpClient == supplied {
t.Fatal("expected constructed client to use a cloned HTTP client")
}
if client.httpClient.Timeout != suppliedTimeout {
t.Fatalf("expected cloned client timeout %v, got %v", suppliedTimeout, client.httpClient.Timeout)
}
if client.httpClient.Transport != transport {
t.Fatal("expected cloned client to preserve the supplied transport")
}
}
func TestOpenAICompatibleClientGenerateSuccess(t *testing.T) { func TestOpenAICompatibleClientGenerateSuccess(t *testing.T) {
type observedRequest struct { type observedRequest struct {
Authorization string Authorization string
@@ -89,6 +148,9 @@ func TestOpenAICompatibleClientGenerateSuccess(t *testing.T) {
if resp.Usage.PromptTokens != 11 || resp.Usage.CompletionTokens != 22 || resp.Usage.TotalTokens != 33 { if resp.Usage.PromptTokens != 11 || resp.Usage.CompletionTokens != 22 || resp.Usage.TotalTokens != 33 {
t.Fatalf("unexpected usage: %+v", resp.Usage) t.Fatalf("unexpected usage: %+v", resp.Usage)
} }
if resp.Usage.CachedTokens != 0 || resp.Usage.CacheWriteTokens != 0 {
t.Fatalf("expected absent cache usage fields to remain zero, got %+v", resp.Usage)
}
if obs.Authorization != "Bearer secret-key" { if obs.Authorization != "Bearer secret-key" {
t.Fatalf("unexpected Authorization header: %q", obs.Authorization) t.Fatalf("unexpected Authorization header: %q", obs.Authorization)
@@ -144,6 +206,277 @@ func TestOpenAICompatibleClientGenerateSuccess(t *testing.T) {
} }
} }
func TestOpenAICompatibleClientDirectAPIKeyPreferredOverEnv(t *testing.T) {
const directKey = "direct-llm-key"
t.Setenv("SCRIPTORIUM_TEST_API_KEY", "env-key")
var gotAuth string
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
gotAuth = r.Header.Get("Authorization")
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{
Model: "model",
APIKeyEnv: "SCRIPTORIUM_TEST_API_KEY",
APIKey: directKey,
},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if gotAuth != "Bearer "+directKey {
t.Fatalf("unexpected Authorization header: %q", gotAuth)
}
}
func TestOpenAICompatibleClientSerializesCacheControlledMessageAsContentBlock(t *testing.T) {
var observedBody map[string]any
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
defer r.Body.Close()
if err := json.NewDecoder(r.Body).Decode(&observedBody); err != nil {
t.Fatalf("failed to decode request body: %v", err)
}
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{
{
Role: "system",
Content: "Stable instructions.",
CacheControl: &domain.CacheControl{
Type: domain.CacheControlEphemeral,
TTL: "1h",
},
},
{Role: "user", Content: "Dynamic request."},
}},
Target: domain.ExecutionTarget{Model: "model"},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
for _, forbidden := range []string{"cache_control", "extra_params"} {
if _, exists := observedBody[forbidden]; exists {
t.Fatalf("expected top-level %s to be omitted, got %#v", forbidden, observedBody[forbidden])
}
}
msgs, ok := observedBody["messages"].([]any)
if !ok || len(msgs) != 2 {
t.Fatalf("unexpected messages payload: %#v", observedBody["messages"])
}
msg0 := msgs[0].(map[string]any)
if msg0["role"] != "system" {
t.Fatalf("unexpected first message role: %#v", msg0["role"])
}
contentBlocks, ok := msg0["content"].([]any)
if !ok || len(contentBlocks) != 1 {
t.Fatalf("expected first message content block array, got %#v", msg0["content"])
}
block := contentBlocks[0].(map[string]any)
if block["type"] != "text" || block["text"] != "Stable instructions." {
t.Fatalf("unexpected text content block: %#v", block)
}
cacheControl, ok := block["cache_control"].(map[string]any)
if !ok {
t.Fatalf("expected cache_control on content block, got %#v", block)
}
if cacheControl["type"] != string(domain.CacheControlEphemeral) || cacheControl["ttl"] != "1h" {
t.Fatalf("unexpected cache_control payload: %#v", cacheControl)
}
msg1 := msgs[1].(map[string]any)
if msg1["role"] != "user" || msg1["content"] != "Dynamic request." {
t.Fatalf("expected uncached message to keep string content, got %#v", msg1)
}
}
func TestOpenAICompatibleClientOmitsEmptyCacheControlTTL(t *testing.T) {
var observedBody map[string]any
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
defer r.Body.Close()
if err := json.NewDecoder(r.Body).Decode(&observedBody); err != nil {
t.Fatalf("failed to decode request body: %v", err)
}
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{
{
Role: "system",
Content: "Stable instructions.",
CacheControl: &domain.CacheControl{
Type: domain.CacheControlEphemeral,
},
},
}},
Target: domain.ExecutionTarget{Model: "model"},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
msgs := observedBody["messages"].([]any)
msg0 := msgs[0].(map[string]any)
contentBlocks := msg0["content"].([]any)
block := contentBlocks[0].(map[string]any)
cacheControl := block["cache_control"].(map[string]any)
if cacheControl["type"] != string(domain.CacheControlEphemeral) {
t.Fatalf("unexpected cache_control type: %#v", cacheControl)
}
if _, exists := cacheControl["ttl"]; exists {
t.Fatalf("expected empty ttl to be omitted, got %#v", cacheControl)
}
}
func TestOpenAICompatibleClientSerializesSessionID(t *testing.T) {
var observedBody map[string]any
var observedSessionHeader string
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
observedSessionHeader = r.Header.Get("x-session-id")
defer r.Body.Close()
if err := json.NewDecoder(r.Body).Decode(&observedBody); err != nil {
t.Fatalf("failed to decode request body: %v", err)
}
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{
SessionID: " session-123 ",
Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}},
},
Target: domain.ExecutionTarget{Model: "model"},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if observedBody["session_id"] != "session-123" {
t.Fatalf("expected top-level session_id, got %#v", observedBody["session_id"])
}
if observedSessionHeader != "" {
t.Fatalf("did not expect x-session-id header, got %q", observedSessionHeader)
}
}
func TestOpenAICompatibleClientOmitsEmptySessionID(t *testing.T) {
var observedBody map[string]any
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
defer r.Body.Close()
if err := json.NewDecoder(r.Body).Decode(&observedBody); err != nil {
t.Fatalf("failed to decode request body: %v", err)
}
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{
SessionID: " ",
Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}},
},
Target: domain.ExecutionTarget{Model: "model"},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if _, exists := observedBody["session_id"]; exists {
t.Fatalf("expected empty session_id to be omitted, got %#v", observedBody["session_id"])
}
}
func TestOpenAICompatibleClientRejectsTooLongSessionID(t *testing.T) {
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{
BaseURL: "http://example.com/v1",
Model: "model",
})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{
SessionID: strings.Repeat("x", domain.SessionIDMaxLength+1),
Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}},
},
})
if err == nil {
t.Fatal("expected invalid request error")
}
if !errors.Is(err, ErrInvalidRequest) {
t.Fatalf("expected ErrInvalidRequest, got %v", err)
}
}
func TestOpenAICompatibleClientParsesCacheUsage(t *testing.T) {
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
_, _ = w.Write([]byte(`{
"choices": [{"message": {"role": "assistant", "content": "ok"}}],
"usage": {
"prompt_tokens": 100,
"completion_tokens": 20,
"total_tokens": 120,
"prompt_tokens_details": {"cached_tokens": 80},
"cache_write_tokens": 60
}
}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1", Model: "model"})
if err != nil {
t.Fatal(err)
}
resp, err := client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if resp.Usage.PromptTokens != 100 || resp.Usage.CompletionTokens != 20 || resp.Usage.TotalTokens != 120 {
t.Fatalf("unexpected base usage fields: %+v", resp.Usage)
}
if resp.Usage.CachedTokens != 80 || resp.Usage.CacheWriteTokens != 60 {
t.Fatalf("unexpected cache usage fields: %+v", resp.Usage)
}
}
func TestOpenAICompatibleClientOmitsResponseFormatWhenNoStructuredOutput(t *testing.T) { func TestOpenAICompatibleClientOmitsResponseFormatWhenNoStructuredOutput(t *testing.T) {
var observedBody map[string]any var observedBody map[string]any
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
@@ -175,7 +508,7 @@ func TestOpenAICompatibleClientOmitsResponseFormatWhenNoStructuredOutput(t *test
} }
} }
func TestOpenAICompatibleClientOmitsReasoningEffortAndExtraParams(t *testing.T) { func TestOpenAICompatibleClientSerializesReasoningEffortAndExtraParams(t *testing.T) {
var observedBody map[string]any var observedBody map[string]any
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
defer r.Body.Close() defer r.Body.Close()
@@ -196,19 +529,257 @@ func TestOpenAICompatibleClientOmitsReasoningEffortAndExtraParams(t *testing.T)
Target: domain.ExecutionTarget{ Target: domain.ExecutionTarget{
Model: "model", Model: "model",
ReasoningEffort: "high", ReasoningEffort: "high",
ExtraParams: map[string]string{ ExtraParams: map[string]any{
"provider_option": "on", "string_value": "on",
"number_value": 42,
"boolean_value": true,
"object_value": map[string]any{"nested": "value", "count": 2},
"array_value": []any{"first", 3, false},
}, },
}, },
}) })
if err != nil { if err != nil {
t.Fatalf("expected no error, got %v", err) t.Fatalf("expected no error, got %v", err)
} }
if observedBody["reasoning_effort"] != "high" {
t.Fatalf("expected reasoning_effort high, got %#v", observedBody["reasoning_effort"])
}
if observedBody["string_value"] != "on" {
t.Fatalf("unexpected string extra param: %#v", observedBody["string_value"])
}
if observedBody["number_value"] != float64(42) {
t.Fatalf("unexpected number extra param: %#v", observedBody["number_value"])
}
if observedBody["boolean_value"] != true {
t.Fatalf("unexpected boolean extra param: %#v", observedBody["boolean_value"])
}
objectValue, ok := observedBody["object_value"].(map[string]any)
if !ok || objectValue["nested"] != "value" || objectValue["count"] != float64(2) {
t.Fatalf("unexpected object extra param: %#v", observedBody["object_value"])
}
if _, exists := observedBody["extra_params"]; exists {
t.Fatalf("expected extra_params wrapper omitted, got %#v", observedBody["extra_params"])
}
arrayValue, ok := observedBody["array_value"].([]any)
if !ok || len(arrayValue) != 3 || arrayValue[0] != "first" || arrayValue[1] != float64(3) || arrayValue[2] != false {
t.Fatalf("unexpected array extra param: %#v", observedBody["array_value"])
}
}
func TestOpenAICompatibleClientOmitsReasoningEffortWhenUnset(t *testing.T) {
var observedBody map[string]any
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
defer r.Body.Close()
if err := json.NewDecoder(r.Body).Decode(&observedBody); err != nil {
t.Fatalf("failed to decode request body: %v", err)
}
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{Model: "model"},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if _, exists := observedBody["reasoning_effort"]; exists { if _, exists := observedBody["reasoning_effort"]; exists {
t.Fatalf("expected reasoning_effort omitted, got %#v", observedBody["reasoning_effort"]) t.Fatalf("expected reasoning_effort omitted, got %#v", observedBody["reasoning_effort"])
} }
if _, exists := observedBody["extra_params"]; exists { if _, exists := observedBody["extra_params"]; exists {
t.Fatalf("expected extra_params omitted, got %#v", observedBody["extra_params"]) t.Fatalf("expected extra_params wrapper omitted, got %#v", observedBody["extra_params"])
}
}
func TestOpenAICompatibleClientSerializesExplicitZeroNumericOverrides(t *testing.T) {
var observedBody map[string]any
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
defer r.Body.Close()
if err := json.NewDecoder(r.Body).Decode(&observedBody); err != nil {
t.Fatalf("failed to decode request body: %v", err)
}
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{Model: "model"},
TargetPresence: domain.ExecutionTargetPresence{
Temperature: true,
MaxTokens: true,
TopP: true,
},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if observedBody["temperature"] != float64(0) {
t.Fatalf("expected explicit zero temperature, got %#v", observedBody["temperature"])
}
if observedBody["max_tokens"] != float64(0) {
t.Fatalf("expected explicit zero max_tokens, got %#v", observedBody["max_tokens"])
}
if observedBody["top_p"] != float64(0) {
t.Fatalf("expected explicit zero top_p, got %#v", observedBody["top_p"])
}
}
func TestOpenAICompatibleClientOmitsImplicitZeroNumericFields(t *testing.T) {
var observedBody map[string]any
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
defer r.Body.Close()
if err := json.NewDecoder(r.Body).Decode(&observedBody); err != nil {
t.Fatalf("failed to decode request body: %v", err)
}
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{Model: "model"},
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
for _, field := range []string{"temperature", "max_tokens", "top_p"} {
if _, exists := observedBody[field]; exists {
t.Fatalf("expected implicit zero field %q to be omitted, got body %#v", field, observedBody)
}
}
}
func TestOpenAICompatibleClientExplicitZeroTimeoutDisablesClientTimeout(t *testing.T) {
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
time.Sleep(20 * time.Millisecond)
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{
BaseURL: ts.URL + "/v1",
Timeout: time.Nanosecond,
})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{Model: "model", TimeoutSeconds: 0},
TargetPresence: domain.ExecutionTargetPresence{TimeoutSeconds: true},
})
if err != nil {
t.Fatalf("expected explicit zero timeout to disable client timeout, got %v", err)
}
}
func TestOpenAICompatibleClientOmittedTimeoutUsesClientTimeout(t *testing.T) {
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
time.Sleep(20 * time.Millisecond)
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{
BaseURL: ts.URL + "/v1",
Timeout: time.Nanosecond,
})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{Model: "model", TimeoutSeconds: 0},
})
if err == nil {
t.Fatal("expected omitted timeout to use client timeout")
}
if !errors.Is(err, ErrRequestFailed) {
t.Fatalf("expected ErrRequestFailed, got %v", err)
}
}
func TestOpenAICompatibleClientRejectsInvalidExtraParamsBeforeProviderCall(t *testing.T) {
tests := []struct {
name string
extraParams map[string]any
want string
}{
{name: "empty key", extraParams: map[string]any{"": "empty"}, want: "key must not be empty"},
{name: "unserializable value", extraParams: map[string]any{"bad": math.Inf(1)}, want: "JSON-serializable"},
}
for _, key := range []string{
"model",
"session_id",
"messages",
"temperature",
"max_tokens",
"top_p",
"service_tier",
"reasoning_effort",
"response_format",
} {
tests = append(tests, struct {
name string
extraParams map[string]any
want string
}{
name: "reserved key " + key,
extraParams: map[string]any{key: "collision"},
want: "reserved request field",
})
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
called := false
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
called = true
_, _ = w.Write([]byte(`{"choices":[{"message":{"content":"ok"}}]}`))
}))
defer ts.Close()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{BaseURL: ts.URL + "/v1"})
if err != nil {
t.Fatal(err)
}
_, err = client.Generate(context.Background(), domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "user", Content: "hi"}}},
Target: domain.ExecutionTarget{Model: "model", ExtraParams: tc.extraParams},
})
if err == nil {
t.Fatal("expected invalid request error")
}
if !errors.Is(err, ErrInvalidRequest) {
t.Fatalf("expected ErrInvalidRequest, got %v", err)
}
if !strings.Contains(err.Error(), tc.want) {
t.Fatalf("expected error to contain %q, got %v", tc.want, err)
}
if called {
t.Fatal("provider should not be called for invalid extra_params")
}
})
} }
} }
@@ -332,9 +903,10 @@ func TestOpenAICompatibleClientEndpointOverride(t *testing.T) {
} }
func TestOpenAICompatibleClientNon2xxError(t *testing.T) { func TestOpenAICompatibleClientNon2xxError(t *testing.T) {
const sensitiveBody = `provider-secret-fragment request_payload_details`
ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { ts := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.WriteHeader(http.StatusBadRequest) w.WriteHeader(http.StatusBadRequest)
_, _ = w.Write([]byte(`{"error":"bad request payload"}`)) _, _ = w.Write([]byte(`{"error":"` + sensitiveBody + `"}`))
})) }))
defer ts.Close() defer ts.Close()
@@ -352,8 +924,11 @@ func TestOpenAICompatibleClientNon2xxError(t *testing.T) {
if !errors.Is(err, ErrUnexpectedStatus) { if !errors.Is(err, ErrUnexpectedStatus) {
t.Fatalf("expected ErrUnexpectedStatus, got %v", err) t.Fatalf("expected ErrUnexpectedStatus, got %v", err)
} }
if !strings.Contains(err.Error(), "400") || !strings.Contains(err.Error(), "bad request payload") { if !strings.Contains(err.Error(), "status=400") {
t.Fatalf("expected status/body details, got %v", err) t.Fatalf("expected status detail, got %v", err)
}
if strings.Contains(err.Error(), sensitiveBody) {
t.Fatalf("expected provider response body to be redacted, got %v", err)
} }
} }

View File

@@ -0,0 +1,9 @@
id: aion-2
endpoint: https://openrouter.ai/api/v1
model: aion-labs/aion-2.0
temperature: 0.72
reasoning_effort: high
top_p: 0.95
timeout_seconds: 180
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,7 @@
id: claude-fable-latest
endpoint: https://openrouter.ai/api/v1
model: "~anthropic/claude-fable-latest"
reasoning_effort: high
timeout_seconds: 600
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,7 @@
id: claude-haiku-latest
endpoint: https://openrouter.ai/api/v1
model: "~anthropic/claude-haiku-latest"
reasoning_effort: medium
timeout_seconds: 240
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,7 @@
id: claude-opus-latest
endpoint: https://openrouter.ai/api/v1
model: "~anthropic/claude-opus-latest"
reasoning_effort: high
timeout_seconds: 240
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,7 @@
id: claude-sonnet-latest
endpoint: https://openrouter.ai/api/v1
model: "~anthropic/claude-sonnet-latest"
reasoning_effort: high
timeout_seconds: 240
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,7 @@
id: deepseek-3-2
endpoint: https://openrouter.ai/api/v1
model: deepseek/deepseek-v3.2
reasoning_effort: high
timeout_seconds: 180
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,7 @@
id: deepseek-4-pro
endpoint: https://openrouter.ai/api/v1
model: deepseek/deepseek-v4-pro
reasoning_effort: high
timeout_seconds: 180
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,9 @@
id: gemini-2-flash-lite
endpoint: https://openrouter.ai/api/v1
model: "google/gemini-2.5-flash-lite"
#temperature: 0.15
reasoning_effort: high
#top_p: 0.98
timeout_seconds: 240
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,9 @@
id: gemini-2-flash
endpoint: https://openrouter.ai/api/v1
model: "google/gemini-2.5-flash"
#temperature: 0.15
reasoning_effort: high
#top_p: 0.98
timeout_seconds: 240
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,9 @@
id: gemini-2-pro
endpoint: https://openrouter.ai/api/v1
model: "google/gemini-2.5-pro"
#temperature: 0.15
reasoning_effort: high
#top_p: 0.98
timeout_seconds: 240
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,9 @@
id: gemini-3-flash-lite
endpoint: https://openrouter.ai/api/v1
model: "google/gemini-3.1-flash-lite"
#temperature: 0.15
reasoning_effort: high
#top_p: 0.98
timeout_seconds: 240
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,9 @@
id: gemini-flash-latest
endpoint: https://openrouter.ai/api/v1
model: "~google/gemini-flash-latest"
#temperature: 0.15
reasoning_effort: high
#top_p: 0.98
timeout_seconds: 240
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,9 @@
id: gemini-pro-latest
endpoint: https://openrouter.ai/api/v1
model: "~google/gemini-pro-latest"
#temperature: 0.15
reasoning_effort: high
#top_p: 0.98
timeout_seconds: 240
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,9 @@
id: gemma-4-31b
endpoint: https://openrouter.ai/api/v1
model: google/gemma-4-31b-it:exacto
temperature: 0.15
reasoning_effort: high
top_p: 0.98
timeout_seconds: 240
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,9 @@
id: minimax-m2
endpoint: https://openrouter.ai/api/v1
model: minimax/minimax-m2.5
temperature: 0.5
reasoning_effort: high
top_p: 0.95
timeout_seconds: 180
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,9 @@
id: minimax-m3
endpoint: https://openrouter.ai/api/v1
model: minimax/minimax-m3
#temperature: 0.5
reasoning_effort: high
#top_p: 0.95
timeout_seconds: 180
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,7 @@
id: mistral-large-2512
endpoint: https://openrouter.ai/api/v1
model: mistralai/mistral-large-2512
temperature: 0.15
top_p: 0.98
timeout_seconds: 180
api_key_env: OPENROUTER_API_KEY

View File

@@ -0,0 +1,8 @@
id: mistral-medium-3-5
endpoint: https://openrouter.ai/api/v1
model: mistralai/mistral-medium-3-5
temperature: 0.15
reasoning_effort: high
top_p: 0.98
timeout_seconds: 180
api_key_env: OPENROUTER_API_KEY

View File

@@ -0,0 +1,7 @@
id: mistral-small-3
endpoint: https://openrouter.ai/api/v1
model: mistralai/mistral-small-3.2-24b-instruct
temperature: 0.05
top_p: 1.0
timeout_seconds: 180
api_key_env: OPENROUTER_API_KEY

View File

@@ -0,0 +1,8 @@
id: mistral-small-4
endpoint: https://openrouter.ai/api/v1
model: mistralai/mistral-small-2603
temperature: 0.1
reasoning_effort: high
top_p: 0.98
timeout_seconds: 180
api_key_env: OPENROUTER_API_KEY

View File

@@ -0,0 +1,7 @@
id: nemotron-3-ultra
endpoint: https://openrouter.ai/api/v1
model: nvidia/nemotron-3-ultra-550b-a55b
reasoning_effort: high
timeout_seconds: 180
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,7 @@
id: gpt-5-mini
endpoint: https://openrouter.ai/api/v1
model: "openai/gpt-5.4-mini"
reasoning_effort: high
timeout_seconds: 240
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,7 @@
id: gpt-5-nano
endpoint: https://openrouter.ai/api/v1
model: "openai/gpt-5.4-nano"
reasoning_effort: high
timeout_seconds: 240
api_key_env: OPENROUTER_API_KEY
service_tier: flex

View File

@@ -0,0 +1,31 @@
package builtin
import (
"embed"
"strings"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile"
)
const assetRoot = "assets"
//go:embed assets/**/*.yml
var assets embed.FS
func NewRepository() profile.Repository {
return profile.NewFSRepository(assets, assetRoot)
}
func NewRepositoryWithPrimary(primary profile.Repository) profile.Repository {
if primary == nil {
return NewRepository()
}
return profile.NewOverlayRepository(primary, NewRepository())
}
func NewRepositoryWithDirectory(dir string) profile.Repository {
if strings.TrimSpace(dir) == "" {
return NewRepository()
}
return NewRepositoryWithPrimary(profile.NewFilesystemRepository(dir))
}

View File

@@ -0,0 +1,127 @@
package builtin
import (
"context"
"errors"
"io/fs"
"strings"
"testing"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile"
"gopkg.in/yaml.v3"
)
func TestBuiltInProfilesValidateThroughRepository(t *testing.T) {
repo := NewRepository()
ids := loadBuiltInProfileIDs(t)
if len(ids) == 0 {
t.Fatal("expected built-in profiles")
}
for id := range ids {
t.Run(id, func(t *testing.T) {
p, err := repo.GetProfile(context.Background(), id)
if err != nil {
t.Fatalf("expected built-in profile %q to load, got %v", id, err)
}
if p.ID != id {
t.Fatalf("expected profile id %q, got %q", id, p.ID)
}
})
}
}
func TestBuiltInProfilesDoNotContainDuplicateIDsOrRawAPIKeys(t *testing.T) {
loadBuiltInProfileIDs(t)
}
func loadBuiltInProfileIDs(t *testing.T) map[string]string {
t.Helper()
ids := map[string]string{}
err := fs.WalkDir(assets, assetRoot, func(name string, d fs.DirEntry, err error) error {
if err != nil {
return err
}
if d.IsDir() || !strings.HasSuffix(name, ".yml") {
return nil
}
data, err := assets.ReadFile(name)
if err != nil {
t.Fatalf("failed to read built-in profile %s: %v", name, err)
}
var raw map[string]any
if err := yaml.Unmarshal(data, &raw); err != nil {
t.Fatalf("failed to decode built-in profile %s: %v", name, err)
}
if _, ok := raw["api_key"]; ok {
t.Fatalf("built-in profile %s contains raw api_key", name)
}
id, ok := raw["id"].(string)
if !ok || strings.TrimSpace(id) == "" {
t.Fatalf("built-in profile %s has missing id", name)
}
if previous, ok := ids[id]; ok {
t.Fatalf("duplicate built-in profile id %q in %s and %s", id, previous, name)
}
ids[id] = name
return nil
})
if err != nil {
t.Fatalf("failed to walk built-in profiles: %v", err)
}
return ids
}
func TestRepositoryWithPrimaryUsesPrimaryBeforeBuiltIns(t *testing.T) {
repo := NewRepositoryWithPrimary(staticProfileRepo{
profiles: map[string]string{"mistral-small-3": "custom-model"},
})
p, err := repo.GetProfile(context.Background(), "mistral-small-3")
if err != nil {
t.Fatalf("expected profile to load, got %v", err)
}
if p.Model != "custom-model" {
t.Fatalf("expected primary profile to override built-in, got %+v", p)
}
}
func TestRepositoryWithPrimaryFallsBackToBuiltIns(t *testing.T) {
repo := NewRepositoryWithPrimary(staticProfileRepo{})
p, err := repo.GetProfile(context.Background(), "mistral-small-3")
if err != nil {
t.Fatalf("expected built-in profile to load, got %v", err)
}
if p.ID != "mistral-small-3" {
t.Fatalf("unexpected profile: %+v", p)
}
}
func TestRepositoryWithPrimaryDoesNotFallBackAfterPrimaryError(t *testing.T) {
repo := NewRepositoryWithPrimary(staticProfileRepo{err: profile.ErrInvalidProfile})
_, err := repo.GetProfile(context.Background(), "mistral-small-3")
if !errors.Is(err, profile.ErrInvalidProfile) {
t.Fatalf("expected primary error, got %v", err)
}
}
type staticProfileRepo struct {
profiles map[string]string
err error
}
func (r staticProfileRepo) GetProfile(_ context.Context, id string) (*domain.ExecutionProfile, error) {
if r.err != nil {
return nil, r.err
}
if model, ok := r.profiles[id]; ok {
return &domain.ExecutionProfile{ID: id, Endpoint: "http://primary/v1", Model: model}, nil
}
return nil, profile.ErrProfileNotFound
}

View File

@@ -5,8 +5,9 @@ import (
"context" "context"
"errors" "errors"
"fmt" "fmt"
"io/fs"
"os" "os"
"path/filepath" "path"
"strings" "strings"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
@@ -30,11 +31,56 @@ func NewFilesystemRepository(dir string) Repository {
} }
func (r *filesystemRepository) GetProfile(ctx context.Context, id string) (*domain.ExecutionProfile, error) { func (r *filesystemRepository) GetProfile(ctx context.Context, id string) (*domain.ExecutionProfile, error) {
return loadProfile(ctx, os.DirFS(r.dir), ".", id)
}
type fsRepository struct {
fsys fs.FS
root string
}
func NewFSRepository(fsys fs.FS, root string) Repository {
return &fsRepository{fsys: fsys, root: root}
}
func (r *fsRepository) GetProfile(ctx context.Context, id string) (*domain.ExecutionProfile, error) {
return loadProfile(ctx, r.fsys, r.root, id)
}
type overlayRepository struct {
primary Repository
fallback Repository
}
func NewOverlayRepository(primary, fallback Repository) Repository {
return &overlayRepository{primary: primary, fallback: fallback}
}
func (r *overlayRepository) GetProfile(ctx context.Context, id string) (*domain.ExecutionProfile, error) {
if r.primary != nil {
prof, err := r.primary.GetProfile(ctx, id)
if err == nil {
return prof, nil
}
if !errors.Is(err, ErrProfileNotFound) {
return nil, err
}
}
if r.fallback == nil {
return nil, ErrProfileNotFound
}
return r.fallback.GetProfile(ctx, id)
}
func loadProfile(ctx context.Context, fsys fs.FS, root string, id string) (*domain.ExecutionProfile, error) {
if strings.TrimSpace(id) == "" { if strings.TrimSpace(id) == "" {
return nil, fmt.Errorf("%w: profile id is required", ErrInvalidProfile) return nil, fmt.Errorf("%w: profile id is required", ErrInvalidProfile)
} }
if fsys == nil {
return nil, fmt.Errorf("failed to read profile directory: filesystem is nil")
}
files, err := filecatalog.FindYAMLFiles(ctx, r.dir) files, err := filecatalog.FindFSYAMLFiles(ctx, fsys, root)
if err != nil { if err != nil {
return nil, fmt.Errorf("failed to read profile directory: %w", err) return nil, fmt.Errorf("failed to read profile directory: %w", err)
} }
@@ -47,24 +93,25 @@ func (r *filesystemRepository) GetProfile(ctx context.Context, id string) (*doma
default: default:
} }
relPath := filecatalog.RelativePath(r.dir, fullPath) relPath := filecatalog.DisplayPath(root, fullPath)
fileMatch := filecatalog.Stem(filepath.Base(fullPath)) == id fileMatch := filecatalog.Stem(path.Base(fullPath)) == id
data, err := os.ReadFile(fullPath) data, err := fs.ReadFile(fsys, fullPath)
if err != nil { if err != nil {
return nil, fmt.Errorf("failed to read profile file %s: %w", relPath, err) return nil, fmt.Errorf("failed to read profile file %s: %w", relPath, err)
} }
metadata := readProfileFileMetadata(data)
idMatch := fileMatch || metadata.id == id
if metadata.hasRawAPIKey {
if idMatch {
return nil, fmt.Errorf("%w: %s", ErrRawAPIKeyNotAllowed, relPath)
}
continue
}
var prof domain.ExecutionProfile var prof domain.ExecutionProfile
decoder := yaml.NewDecoder(bytes.NewReader(data)) decoder := yaml.NewDecoder(bytes.NewReader(data))
decoder.KnownFields(true) decoder.KnownFields(true)
if err := decoder.Decode(&prof); err != nil { if err := decoder.Decode(&prof); err != nil {
idMatch := fileMatch || profileFileHasID(data, id)
if strings.Contains(err.Error(), "field api_key not found") {
if idMatch {
return nil, fmt.Errorf("%w: %s", ErrRawAPIKeyNotAllowed, relPath)
}
continue
}
if idMatch { if idMatch {
return nil, fmt.Errorf("%w: %s: %v", ErrInvalidYAML, relPath, err) return nil, fmt.Errorf("%w: %s: %v", ErrInvalidYAML, relPath, err)
} }
@@ -106,14 +153,36 @@ type profileMatch struct {
path string path string
} }
func profileFileHasID(data []byte, id string) bool { type profileFileMetadata struct {
var raw struct { id string
ID string `yaml:"id"` hasRawAPIKey bool
}
func readProfileFileMetadata(data []byte) profileFileMetadata {
var node yaml.Node
if err := yaml.NewDecoder(bytes.NewReader(data)).Decode(&node); err != nil {
return profileFileMetadata{}
} }
if err := yaml.NewDecoder(bytes.NewReader(data)).Decode(&raw); err != nil { if node.Kind != yaml.DocumentNode || len(node.Content) == 0 {
return false return profileFileMetadata{}
} }
return strings.TrimSpace(raw.ID) == id mapping := node.Content[0]
if mapping.Kind != yaml.MappingNode {
return profileFileMetadata{}
}
var metadata profileFileMetadata
for i := 0; i+1 < len(mapping.Content); i += 2 {
key := mapping.Content[i]
value := mapping.Content[i+1]
switch key.Value {
case "id":
metadata.id = strings.TrimSpace(value.Value)
case "api_key":
metadata.hasRawAPIKey = true
}
}
return metadata
} }
func validateProfile(p *domain.ExecutionProfile) error { func validateProfile(p *domain.ExecutionProfile) error {

View File

@@ -2,11 +2,15 @@ package profile
import ( import (
"context" "context"
"encoding/json"
"errors" "errors"
"os" "os"
"path/filepath" "path/filepath"
"strings" "strings"
"testing" "testing"
"testing/fstest"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
) )
func TestFilesystemRepository_GetProfile(t *testing.T) { func TestFilesystemRepository_GetProfile(t *testing.T) {
@@ -85,6 +89,63 @@ temperature: 0.1
} }
}) })
t.Run("valid profile with JSON-compatible extra params", func(t *testing.T) {
writeProfileTestFile(t, filepath.Join(tmpDir, "json-extra-params.yaml"), `
id: json-extra-params
endpoint: http://localhost:8000/v1
model: nested-model
extra_params:
string_value: enabled
number_value: 42
boolean_value: true
object_value:
nested: value
count: 2
array_value:
- first
- 3
- false
`)
p, err := repo.GetProfile(ctx, "json-extra-params")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
var got map[string]any
encoded, err := json.Marshal(p.ExtraParams)
if err != nil {
t.Fatalf("expected extra_params to marshal as JSON, got %v", err)
}
if err := json.Unmarshal(encoded, &got); err != nil {
t.Fatalf("expected extra_params JSON to decode, got %v", err)
}
if got["string_value"] != "enabled" {
t.Fatalf("unexpected string extra param: %#v", got["string_value"])
}
if got["number_value"] != float64(42) {
t.Fatalf("unexpected number extra param: %#v", got["number_value"])
}
if got["boolean_value"] != true {
t.Fatalf("unexpected boolean extra param: %#v", got["boolean_value"])
}
objectValue, ok := got["object_value"].(map[string]any)
if !ok {
t.Fatalf("expected object extra param, got %#v", got["object_value"])
}
if objectValue["nested"] != "value" || objectValue["count"] != float64(2) {
t.Fatalf("unexpected object extra param: %#v", objectValue)
}
arrayValue, ok := got["array_value"].([]any)
if !ok {
t.Fatalf("expected array extra param, got %#v", got["array_value"])
}
if len(arrayValue) != 3 || arrayValue[0] != "first" || arrayValue[1] != float64(3) || arrayValue[2] != false {
t.Fatalf("unexpected array extra param: %#v", arrayValue)
}
})
t.Run("duplicate profile IDs fail as ambiguous", func(t *testing.T) { t.Run("duplicate profile IDs fail as ambiguous", func(t *testing.T) {
writeProfileTestFile(t, filepath.Join(tmpDir, "duplicate-profile-a.yaml"), ` writeProfileTestFile(t, filepath.Join(tmpDir, "duplicate-profile-a.yaml"), `
id: duplicate-profile id: duplicate-profile
@@ -133,6 +194,20 @@ api_key: secret
} }
}) })
t.Run("raw api_key in non-target profile is ignored", func(t *testing.T) {
writeProfileTestFile(t, filepath.Join(tmpDir, "raw-api-key-non-target.yaml"), `
id: raw-api-key-non-target
endpoint: http://localhost:8000/v1
model: m
api_key: secret
`)
_, err := repo.GetProfile(ctx, "does-not-exist-with-raw-key-nearby")
if !errors.Is(err, ErrProfileNotFound) {
t.Fatalf("expected ErrProfileNotFound for non-target raw api_key file, got %v", err)
}
})
t.Run("invalid yaml", func(t *testing.T) { t.Run("invalid yaml", func(t *testing.T) {
_, err := repo.GetProfile(ctx, "invalid_yaml") _, err := repo.GetProfile(ctx, "invalid_yaml")
if !errors.Is(err, ErrInvalidYAML) { if !errors.Is(err, ErrInvalidYAML) {
@@ -189,3 +264,216 @@ func writeProfileTestFile(t *testing.T, path string, content string) {
t.Fatalf("failed to write profile test file %q: %v", path, err) t.Fatalf("failed to write profile test file %q: %v", path, err)
} }
} }
func TestFSRepository(t *testing.T) {
ctx := context.Background()
t.Run("loads valid profiles from nested directories", func(t *testing.T) {
repo := NewFSRepository(fstest.MapFS{
"profiles/provider/nested.yaml": profileMapFile(`
id: nested-profile
endpoint: http://localhost:8000/v1
model: nested-model
temperature: 0.1
`),
}, "profiles")
p, err := repo.GetProfile(ctx, "nested-profile")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if p.ID != "nested-profile" || p.Model != "nested-model" {
t.Fatalf("unexpected profile: %+v", p)
}
})
t.Run("rejects unknown YAML fields", func(t *testing.T) {
repo := NewFSRepository(fstest.MapFS{
"profiles/unknown.yaml": profileMapFile(`
id: unknown-profile
endpoint: http://localhost:8000/v1
model: model
unknown: value
`),
}, "profiles")
_, err := repo.GetProfile(ctx, "unknown-profile")
if !errors.Is(err, ErrInvalidYAML) {
t.Fatalf("expected ErrInvalidYAML, got %v", err)
}
})
t.Run("rejects raw api_key in selected profile", func(t *testing.T) {
repo := NewFSRepository(fstest.MapFS{
"profiles/raw.yaml": profileMapFile(`
id: raw-profile
endpoint: http://localhost:8000/v1
model: model
api_key: secret
`),
}, "profiles")
_, err := repo.GetProfile(ctx, "raw-profile")
if !errors.Is(err, ErrRawAPIKeyNotAllowed) {
t.Fatalf("expected ErrRawAPIKeyNotAllowed, got %v", err)
}
})
t.Run("ignores raw api_key in non-selected profiles", func(t *testing.T) {
repo := NewFSRepository(fstest.MapFS{
"profiles/raw.yaml": profileMapFile(`
id: raw-profile
endpoint: http://localhost:8000/v1
model: model
api_key: secret
`),
"profiles/valid.yaml": profileMapFile(`
id: valid-profile
endpoint: http://localhost:8000/v1
model: model
`),
}, "profiles")
p, err := repo.GetProfile(ctx, "valid-profile")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if p.ID != "valid-profile" {
t.Fatalf("unexpected profile: %+v", p)
}
})
t.Run("rejects duplicate IDs within one source", func(t *testing.T) {
repo := NewFSRepository(fstest.MapFS{
"profiles/a.yaml": profileMapFile(`
id: duplicate-profile
endpoint: http://localhost:8000/v1
model: first
`),
"profiles/nested/b.yaml": profileMapFile(`
id: duplicate-profile
endpoint: http://localhost:8000/v1
model: second
`),
}, "profiles")
_, err := repo.GetProfile(ctx, "duplicate-profile")
if !errors.Is(err, ErrInvalidProfile) {
t.Fatalf("expected ErrInvalidProfile, got %v", err)
}
for _, want := range []string{"duplicate execution profile id", "a.yaml", "nested/b.yaml"} {
if !strings.Contains(err.Error(), want) {
t.Fatalf("expected error to contain %q, got %v", want, err)
}
}
})
}
func TestOverlayRepository(t *testing.T) {
ctx := context.Background()
primaryProfile := &domain.ExecutionProfile{ID: "shared", Endpoint: "http://primary", Model: "primary"}
fallbackProfile := &domain.ExecutionProfile{ID: "shared", Endpoint: "http://fallback", Model: "fallback"}
t.Run("returns primary matches before fallback matches", func(t *testing.T) {
repo := NewOverlayRepository(
staticProfileRepo{profiles: map[string]*domain.ExecutionProfile{"shared": primaryProfile}},
staticProfileRepo{profiles: map[string]*domain.ExecutionProfile{"shared": fallbackProfile}},
)
p, err := repo.GetProfile(ctx, "shared")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if p.Model != "primary" {
t.Fatalf("expected primary profile, got %+v", p)
}
})
t.Run("falls back on primary not found", func(t *testing.T) {
repo := NewOverlayRepository(
staticProfileRepo{},
staticProfileRepo{profiles: map[string]*domain.ExecutionProfile{"shared": fallbackProfile}},
)
p, err := repo.GetProfile(ctx, "shared")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if p.Model != "fallback" {
t.Fatalf("expected fallback profile, got %+v", p)
}
})
t.Run("does not fall back after primary load errors", func(t *testing.T) {
for _, tc := range []struct {
name string
err error
}{
{name: "invalid yaml", err: ErrInvalidYAML},
{name: "invalid profile", err: ErrInvalidProfile},
{name: "raw api key", err: ErrRawAPIKeyNotAllowed},
} {
t.Run(tc.name, func(t *testing.T) {
repo := NewOverlayRepository(
staticProfileRepo{err: tc.err},
staticProfileRepo{profiles: map[string]*domain.ExecutionProfile{"shared": fallbackProfile}},
)
_, err := repo.GetProfile(ctx, "shared")
if !errors.Is(err, tc.err) {
t.Fatalf("expected %v, got %v", tc.err, err)
}
})
}
})
t.Run("returns not found when both sources miss", func(t *testing.T) {
repo := NewOverlayRepository(staticProfileRepo{}, staticProfileRepo{})
_, err := repo.GetProfile(ctx, "missing")
if !errors.Is(err, ErrProfileNotFound) {
t.Fatalf("expected ErrProfileNotFound, got %v", err)
}
})
t.Run("nil primary uses fallback", func(t *testing.T) {
repo := NewOverlayRepository(nil, staticProfileRepo{profiles: map[string]*domain.ExecutionProfile{"shared": fallbackProfile}})
p, err := repo.GetProfile(ctx, "shared")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if p.Model != "fallback" {
t.Fatalf("expected fallback profile, got %+v", p)
}
})
t.Run("nil fallback returns not found after primary miss", func(t *testing.T) {
repo := NewOverlayRepository(staticProfileRepo{}, nil)
_, err := repo.GetProfile(ctx, "missing")
if !errors.Is(err, ErrProfileNotFound) {
t.Fatalf("expected ErrProfileNotFound, got %v", err)
}
})
}
func profileMapFile(content string) *fstest.MapFile {
return &fstest.MapFile{Data: []byte(strings.TrimLeft(content, "\n"))}
}
type staticProfileRepo struct {
profiles map[string]*domain.ExecutionProfile
err error
}
func (r staticProfileRepo) GetProfile(_ context.Context, id string) (*domain.ExecutionProfile, error) {
if r.err != nil {
return nil, r.err
}
if p, ok := r.profiles[id]; ok {
cp := *p
return &cp, nil
}
return nil, ErrProfileNotFound
}

View File

@@ -6,7 +6,9 @@ import (
"errors" "errors"
"fmt" "fmt"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
"strings"
"text/template" "text/template"
"unicode/utf8"
) )
var ( var (
@@ -50,6 +52,11 @@ func (r *goRenderer) Render(ctx context.Context, definition *domain.PromptDefini
}, },
} }
sessionID, err := renderSessionID(definition.SessionID, funcs, vars)
if err != nil {
return nil, err
}
var renderedMessages []domain.RenderedMessage var renderedMessages []domain.RenderedMessage
for i, tmplMsg := range definition.Templates { for i, tmplMsg := range definition.Templates {
@@ -75,12 +82,44 @@ func (r *goRenderer) Render(ctx context.Context, definition *domain.PromptDefini
} }
renderedMessages = append(renderedMessages, domain.RenderedMessage{ renderedMessages = append(renderedMessages, domain.RenderedMessage{
Role: tmplMsg.Role, Role: tmplMsg.Role,
Content: buf.String(), Content: buf.String(),
CacheControl: cloneCacheControl(tmplMsg.CacheControl),
}) })
} }
return &domain.RenderedPrompt{ return &domain.RenderedPrompt{
Messages: renderedMessages, SessionID: sessionID,
Messages: renderedMessages,
}, nil }, nil
} }
func renderSessionID(raw string, funcs template.FuncMap, vars map[string]string) (string, error) {
if strings.TrimSpace(raw) == "" {
return "", nil
}
tmpl, err := template.New("session_id").Funcs(funcs).Option("missingkey=error").Parse(raw)
if err != nil {
return "", fmt.Errorf("%w: session_id: %v", ErrInvalidTemplate, err)
}
var buf bytes.Buffer
if err := tmpl.Execute(&buf, vars); err != nil {
return "", fmt.Errorf("%w: session_id: %w", ErrRenderFailure, err)
}
sessionID := strings.TrimSpace(buf.String())
if n := utf8.RuneCountInString(sessionID); n > domain.SessionIDMaxLength {
return "", fmt.Errorf("%w: session_id length %d exceeds maximum %d", ErrRenderFailure, n, domain.SessionIDMaxLength)
}
return sessionID, nil
}
func cloneCacheControl(in *domain.CacheControl) *domain.CacheControl {
if in == nil {
return nil
}
out := *in
return &out
}

View File

@@ -3,6 +3,7 @@ package prompt
import ( import (
"context" "context"
"errors" "errors"
"strings"
"testing" "testing"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
@@ -78,6 +79,66 @@ func TestGoRenderer_Render(t *testing.T) {
} }
}) })
t.Run("copying cache control to rendered messages", func(t *testing.T) {
def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{
Role: "system",
Content: "You are concise.",
CacheControl: &domain.CacheControl{
Type: domain.CacheControlEphemeral,
TTL: "1h",
},
},
{Role: "user", Content: "Analyze this: {{input \"transcript\"}}"},
},
}
res, err := renderer.Render(ctx, def, inputs, vars)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if len(res.Messages) != 2 {
t.Fatalf("expected 2 messages, got %d", len(res.Messages))
}
if res.Messages[0].CacheControl == nil {
t.Fatal("expected rendered cache control")
}
if res.Messages[0].CacheControl.Type != domain.CacheControlEphemeral {
t.Fatalf("unexpected cache control type: %q", res.Messages[0].CacheControl.Type)
}
if res.Messages[0].CacheControl.TTL != "1h" {
t.Fatalf("unexpected cache control ttl: %q", res.Messages[0].CacheControl.TTL)
}
if res.Messages[1].CacheControl != nil {
t.Fatalf("expected no cache control on second message, got %#v", res.Messages[1].CacheControl)
}
})
t.Run("rendered cache control does not alias source template", func(t *testing.T) {
source := &domain.CacheControl{Type: domain.CacheControlEphemeral, TTL: "1h"}
def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "system", Content: "You are concise.", CacheControl: source},
},
}
res, err := renderer.Render(ctx, def, inputs, vars)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if res.Messages[0].CacheControl == source {
t.Fatal("expected rendered cache control to be cloned")
}
res.Messages[0].CacheControl.TTL = ""
if source.TTL != "1h" {
t.Fatalf("source cache control was mutated, ttl=%q", source.TTL)
}
})
t.Run("accessing vars", func(t *testing.T) { t.Run("accessing vars", func(t *testing.T) {
def := &domain.PromptDefinition{ def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}}, Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
@@ -95,6 +156,78 @@ func TestGoRenderer_Render(t *testing.T) {
} }
}) })
t.Run("rendering session id from vars", func(t *testing.T) {
def := &domain.PromptDefinition{
SessionID: " {{ .session_id }} ",
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "system", Content: "Speak in a {{.tone}} tone."},
},
}
res, err := renderer.Render(ctx, def, inputs, map[string]string{
"tone": "concise",
"session_id": "agent-session-123",
})
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if res.SessionID != "agent-session-123" {
t.Fatalf("unexpected session id: %q", res.SessionID)
}
})
t.Run("empty rendered session id is omitted", func(t *testing.T) {
def := &domain.PromptDefinition{
SessionID: " ",
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "system", Content: "Speak in a {{.tone}} tone."},
},
}
res, err := renderer.Render(ctx, def, inputs, vars)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if res.SessionID != "" {
t.Fatalf("expected empty session id, got %q", res.SessionID)
}
})
t.Run("missing session id var fails rendering", func(t *testing.T) {
def := &domain.PromptDefinition{
SessionID: "{{ .session_id }}",
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "system", Content: "Speak in a {{.tone}} tone."},
},
}
_, err := renderer.Render(ctx, def, inputs, vars)
if !errors.Is(err, ErrRenderFailure) {
t.Fatalf("expected ErrRenderFailure, got %v", err)
}
})
t.Run("too long rendered session id fails rendering", func(t *testing.T) {
def := &domain.PromptDefinition{
SessionID: "{{ .session_id }}",
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},
Templates: []domain.PromptMessageTemplate{
{Role: "system", Content: "Speak in a {{.tone}} tone."},
},
}
_, err := renderer.Render(ctx, def, inputs, map[string]string{
"tone": "concise",
"session_id": strings.Repeat("x", domain.SessionIDMaxLength+1),
})
if !errors.Is(err, ErrRenderFailure) {
t.Fatalf("expected ErrRenderFailure, got %v", err)
}
})
t.Run("inserting required input artifact", func(t *testing.T) { t.Run("inserting required input artifact", func(t *testing.T) {
def := &domain.PromptDefinition{ def := &domain.PromptDefinition{
Inputs: []domain.PromptInput{{Name: "transcript", Required: true}}, Inputs: []domain.PromptInput{{Name: "transcript", Required: true}},

View File

@@ -5,7 +5,9 @@ import (
"context" "context"
"errors" "errors"
"fmt" "fmt"
"io/fs"
"os" "os"
"path"
"path/filepath" "path/filepath"
"strings" "strings"
@@ -24,11 +26,17 @@ type filesystemRepository struct {
dir string dir string
} }
type fsRepository struct {
fsys fs.FS
root string
}
type promptDefinitionFile struct { type promptDefinitionFile struct {
ID string `yaml:"id"` ID string `yaml:"id"`
Version string `yaml:"version"` Version string `yaml:"version"`
DefaultProfile *string `yaml:"default_profile"` DefaultProfile *string `yaml:"default_profile"`
Description string `yaml:"description"` Description string `yaml:"description"`
SessionID string `yaml:"session_id"`
Inputs []promptInputFile `yaml:"inputs"` Inputs []promptInputFile `yaml:"inputs"`
Messages []promptMessageFile `yaml:"messages"` Messages []promptMessageFile `yaml:"messages"`
Output promptOutputContractFile `yaml:"output"` Output promptOutputContractFile `yaml:"output"`
@@ -42,9 +50,15 @@ type promptInputFile struct {
} }
type promptMessageFile struct { type promptMessageFile struct {
Role string `yaml:"role"` Role string `yaml:"role"`
Content string `yaml:"content"` Content string `yaml:"content"`
ContentFile string `yaml:"content_file"` ContentFile string `yaml:"content_file"`
CacheControl *cacheControlFile `yaml:"cache_control"`
}
type cacheControlFile struct {
Type string `yaml:"type"`
TTL string `yaml:"ttl"`
} }
type promptOutputContractFile struct { type promptOutputContractFile struct {
@@ -58,6 +72,10 @@ func NewFilesystemRepository(dir string) Repository {
return &filesystemRepository{dir: dir} return &filesystemRepository{dir: dir}
} }
func NewFSRepository(fsys fs.FS, root string) Repository {
return &fsRepository{fsys: fsys, root: root}
}
func (r *filesystemRepository) GetPromptDefinition(ctx context.Context, id string, version string) (*domain.PromptDefinition, error) { func (r *filesystemRepository) GetPromptDefinition(ctx context.Context, id string, version string) (*domain.PromptDefinition, error) {
if strings.TrimSpace(id) == "" { if strings.TrimSpace(id) == "" {
return nil, fmt.Errorf("%w: prompt id is required", ErrInvalidPromptDefinition) return nil, fmt.Errorf("%w: prompt id is required", ErrInvalidPromptDefinition)
@@ -125,6 +143,10 @@ func (r *filesystemRepository) GetPromptDefinition(ctx context.Context, id strin
return nil, ErrPromptDefinitionNotFound return nil, ErrPromptDefinitionNotFound
} }
func (r *fsRepository) GetPromptDefinition(ctx context.Context, id string, version string) (*domain.PromptDefinition, error) {
return loadPromptDefinition(ctx, r.fsys, r.root, id, version)
}
type promptDefinitionMatch struct { type promptDefinitionMatch struct {
def *domain.PromptDefinition def *domain.PromptDefinition
path string path string
@@ -159,7 +181,152 @@ func promptDefinitionFileHasID(path string, id string) bool {
return strings.TrimSpace(raw.ID) == id return strings.TrimSpace(raw.ID) == id
} }
func loadPromptDefinition(ctx context.Context, fsys fs.FS, root string, id string, version string) (*domain.PromptDefinition, error) {
if strings.TrimSpace(id) == "" {
return nil, fmt.Errorf("%w: prompt id is required", ErrInvalidPromptDefinition)
}
if fsys == nil {
return nil, fmt.Errorf("failed to read prompt definition directory: filesystem is nil")
}
files, err := filecatalog.FindFSYAMLFiles(ctx, fsys, root)
if err != nil {
return nil, fmt.Errorf("failed to read prompt definition directory: %w", err)
}
cleanRoot := filecatalog.CleanFSRoot(root)
rootInfo, err := fs.Stat(fsys, cleanRoot)
if err != nil {
return nil, fmt.Errorf("failed to read prompt definition directory: %w", err)
}
var matches []promptDefinitionMatch
for _, fullPath := range files {
select {
case <-ctx.Done():
return nil, ctx.Err()
default:
}
relPath := filecatalog.DisplayPath(root, fullPath)
fileMatch := filecatalog.Stem(path.Base(fullPath)) == id
data, err := fs.ReadFile(fsys, fullPath)
if err != nil {
if fileMatch {
return nil, fmt.Errorf("%w: %s: failed to read prompt definition file: %v", ErrInvalidYAML, relPath, err)
}
continue
}
raw, err := decodePromptDefinition(data)
if err != nil {
if fileMatch || promptDefinitionDataHasID(data, id) {
return nil, fmt.Errorf("%w: %s: %v", ErrInvalidYAML, relPath, err)
}
continue
}
def, err := normalizePromptDefinitionFromFS(raw, fsys, root, fullPath, rootInfo.IsDir())
if err != nil {
if fileMatch || strings.TrimSpace(raw.ID) == id {
return nil, fmt.Errorf("%w: %s: %v", ErrInvalidPromptDefinition, relPath, err)
}
continue
}
if def.ID != id {
continue
}
if version != "" && def.Version != version {
continue
}
matches = append(matches, promptDefinitionMatch{
def: def,
path: relPath,
})
}
if len(matches) > 1 {
paths := make([]string, 0, len(matches))
for _, match := range matches {
paths = append(paths, match.path)
}
if version != "" {
return nil, fmt.Errorf("%w: duplicate prompt definition id %q version %q found in: %s", ErrInvalidPromptDefinition, id, version, strings.Join(paths, ", "))
}
return nil, fmt.Errorf("%w: duplicate prompt definition id %q found in: %s", ErrInvalidPromptDefinition, id, strings.Join(paths, ", "))
}
if len(matches) == 1 {
return matches[0].def, nil
}
return nil, ErrPromptDefinitionNotFound
}
func decodePromptDefinition(data []byte) (*promptDefinitionFile, error) {
var raw promptDefinitionFile
decoder := yaml.NewDecoder(bytes.NewReader(data))
decoder.KnownFields(true)
if err := decoder.Decode(&raw); err != nil {
return nil, err
}
return &raw, nil
}
func promptDefinitionDataHasID(data []byte, id string) bool {
var raw struct {
ID string `yaml:"id"`
}
if err := yaml.NewDecoder(bytes.NewReader(data)).Decode(&raw); err != nil {
return false
}
return strings.TrimSpace(raw.ID) == id
}
func normalizePromptDefinition(raw *promptDefinitionFile, sourcePath string) (*domain.PromptDefinition, error) { func normalizePromptDefinition(raw *promptDefinitionFile, sourcePath string) (*domain.PromptDefinition, error) {
promptDir := filepath.Dir(sourcePath)
return normalizePromptDefinitionWithContent(raw, func(contentFile string) (string, string, error) {
resolvedPath := strings.TrimSpace(contentFile)
if !filepath.IsAbs(resolvedPath) {
resolvedPath = filepath.Join(promptDir, resolvedPath)
}
resolvedPath = filepath.Clean(resolvedPath)
body, err := os.ReadFile(resolvedPath)
if err != nil {
return "", "", err
}
return string(body), resolvedPath, nil
})
}
func normalizePromptDefinitionFromFS(raw *promptDefinitionFile, fsys fs.FS, root string, sourcePath string, rootIsDir bool) (*domain.PromptDefinition, error) {
promptDir := path.Dir(sourcePath)
return normalizePromptDefinitionWithContent(raw, func(contentFile string) (string, string, error) {
var resolvedPath string
if rootIsDir {
var err error
resolvedPath, _, err = filecatalog.ResolveFSPath(root, promptDir, contentFile)
if err != nil {
return "", "", err
}
} else {
resolvedPath = strings.TrimSpace(contentFile)
if !path.IsAbs(resolvedPath) {
resolvedPath = path.Join(promptDir, resolvedPath)
}
resolvedPath = strings.TrimPrefix(path.Clean(resolvedPath), "/")
}
body, err := fs.ReadFile(fsys, resolvedPath)
if err != nil {
return "", "", err
}
return string(body), resolvedPath, nil
})
}
func normalizePromptDefinitionWithContent(raw *promptDefinitionFile, readContentFile func(string) (string, string, error)) (*domain.PromptDefinition, error) {
if raw == nil { if raw == nil {
return nil, errors.New("prompt definition is nil") return nil, errors.New("prompt definition is nil")
} }
@@ -199,7 +366,6 @@ func normalizePromptDefinition(raw *promptDefinitionFile, sourcePath string) (*d
} }
templates := make([]domain.PromptMessageTemplate, 0, len(raw.Messages)) templates := make([]domain.PromptMessageTemplate, 0, len(raw.Messages))
promptDir := filepath.Dir(sourcePath)
for i, msg := range raw.Messages { for i, msg := range raw.Messages {
role := strings.TrimSpace(msg.Role) role := strings.TrimSpace(msg.Role)
if role == "" { if role == "" {
@@ -212,27 +378,27 @@ func normalizePromptDefinition(raw *promptDefinitionFile, sourcePath string) (*d
return nil, fmt.Errorf("message %d (%s) must set exactly one of content or content_file", i, role) return nil, fmt.Errorf("message %d (%s) must set exactly one of content or content_file", i, role)
} }
cacheControl, err := normalizeCacheControl(msg.CacheControl)
if err != nil {
return nil, fmt.Errorf("message %d (%s) cache_control: %w", i, role, err)
}
templateContent := msg.Content templateContent := msg.Content
resolvedContentFile := "" resolvedContentFile := ""
if hasContentFile { if hasContentFile {
resolvedPath := strings.TrimSpace(msg.ContentFile) body, resolvedPath, err := readContentFile(msg.ContentFile)
if !filepath.IsAbs(resolvedPath) {
resolvedPath = filepath.Join(promptDir, resolvedPath)
}
resolvedPath = filepath.Clean(resolvedPath)
body, err := os.ReadFile(resolvedPath)
if err != nil { if err != nil {
return nil, fmt.Errorf("prompt %q message %d (%s): failed to read content_file %q: %w", id, i, role, msg.ContentFile, err) return nil, fmt.Errorf("prompt %q message %d (%s): failed to read content_file %q: %w", id, i, role, msg.ContentFile, err)
} }
templateContent = string(body) templateContent = body
resolvedContentFile = resolvedPath resolvedContentFile = resolvedPath
} }
templates = append(templates, domain.PromptMessageTemplate{ templates = append(templates, domain.PromptMessageTemplate{
Role: role, Role: role,
Content: templateContent, Content: templateContent,
ContentFile: resolvedContentFile, ContentFile: resolvedContentFile,
CacheControl: cacheControl,
}) })
} }
@@ -262,6 +428,7 @@ func normalizePromptDefinition(raw *promptDefinitionFile, sourcePath string) (*d
Version: version, Version: version,
DefaultProfile: defaultProfile, DefaultProfile: defaultProfile,
Description: strings.TrimSpace(raw.Description), Description: strings.TrimSpace(raw.Description),
SessionID: strings.TrimSpace(raw.SessionID),
Inputs: inputs, Inputs: inputs,
Templates: templates, Templates: templates,
OutputFormat: raw.Output.Format, OutputFormat: raw.Output.Format,
@@ -274,6 +441,30 @@ func normalizePromptDefinition(raw *promptDefinitionFile, sourcePath string) (*d
}, nil }, nil
} }
func normalizeCacheControl(raw *cacheControlFile) (*domain.CacheControl, error) {
if raw == nil {
return nil, nil
}
cacheType := strings.TrimSpace(raw.Type)
if cacheType == "" {
return nil, errors.New("type is required")
}
if domain.CacheControlType(cacheType) != domain.CacheControlEphemeral {
return nil, fmt.Errorf("unsupported type %q", cacheType)
}
ttl := strings.TrimSpace(raw.TTL)
if ttl != "" && ttl != "1h" {
return nil, fmt.Errorf("unsupported ttl %q", ttl)
}
return &domain.CacheControl{
Type: domain.CacheControlType(cacheType),
TTL: ttl,
}, nil
}
func isValidOutputFormat(f domain.OutputFormat) bool { func isValidOutputFormat(f domain.OutputFormat) bool {
switch f { switch f {
case domain.FormatText, domain.FormatMarkdown, domain.FormatJSON: case domain.FormatText, domain.FormatMarkdown, domain.FormatJSON:

View File

@@ -8,6 +8,7 @@ import (
"path/filepath" "path/filepath"
"strings" "strings"
"testing" "testing"
"testing/fstest"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
) )
@@ -68,6 +69,44 @@ func TestFilesystemRepository_GetPromptDefinition(t *testing.T) {
} }
}) })
t.Run("valid cache control with ttl", func(t *testing.T) {
p, err := repo.GetPromptDefinition(ctx, "valid-cache-control-ttl", "")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if len(p.Templates) != 2 {
t.Fatalf("expected 2 messages, got %d", len(p.Templates))
}
assertCacheControl(t, p.Templates[0].CacheControl, domain.CacheControlEphemeral, "1h")
if p.Templates[1].CacheControl != nil {
t.Fatalf("expected second message cache control to be nil, got %#v", p.Templates[1].CacheControl)
}
})
t.Run("valid cache control without ttl", func(t *testing.T) {
p, err := repo.GetPromptDefinition(ctx, "valid-cache-control-without-ttl", "")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if len(p.Templates) != 2 {
t.Fatalf("expected 2 messages, got %d", len(p.Templates))
}
assertCacheControl(t, p.Templates[0].CacheControl, domain.CacheControlEphemeral, "")
if p.Templates[1].CacheControl != nil {
t.Fatalf("expected second message cache control to be nil, got %#v", p.Templates[1].CacheControl)
}
})
t.Run("valid session id template", func(t *testing.T) {
p, err := repo.GetPromptDefinition(ctx, "valid-session-id", "")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if p.SessionID != "{{ .session_id }}" {
t.Fatalf("expected trimmed session_id template, got %q", p.SessionID)
}
})
t.Run("valid nested file-backed prompt resolves content file relative to nested YAML", func(t *testing.T) { t.Run("valid nested file-backed prompt resolves content file relative to nested YAML", func(t *testing.T) {
nestedDir := filepath.Join(tmpDir, "dnd", "recap") nestedDir := filepath.Join(tmpDir, "dnd", "recap")
if err := os.MkdirAll(nestedDir, 0o755); err != nil { if err := os.MkdirAll(nestedDir, 0o755); err != nil {
@@ -258,6 +297,10 @@ output:
{name: "invalid validation mode", id: "invalid_validation_mode", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"invalid validation mode"}}, {name: "invalid validation mode", id: "invalid_validation_mode", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"invalid validation mode"}},
{name: "json_schema without schema_path", id: "json_schema_without_schema_path", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"schema_path"}}, {name: "json_schema without schema_path", id: "json_schema_without_schema_path", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"schema_path"}},
{name: "unknown input field", id: "unknown_input_field", targetErr: ErrInvalidYAML, errSubstrs: []string{"field unknown_input_setting not found"}}, {name: "unknown input field", id: "unknown_input_field", targetErr: ErrInvalidYAML, errSubstrs: []string{"field unknown_input_setting not found"}},
{name: "empty cache control type", id: "empty_cache_control_type", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"cache_control", "type is required"}},
{name: "unsupported cache control type", id: "unsupported_cache_control_type", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"cache_control", "unsupported type"}},
{name: "unsupported cache control ttl", id: "unsupported_cache_control_ttl", targetErr: ErrInvalidPromptDefinition, errSubstrs: []string{"cache_control", "unsupported ttl"}},
{name: "unknown cache control field", id: "unknown_cache_control_field", targetErr: ErrInvalidYAML, errSubstrs: []string{"field unexpected not found"}},
} }
for _, tc := range cases { for _, tc := range cases {
@@ -282,6 +325,173 @@ output:
}) })
} }
func TestFSRepositoryGetPromptDefinition(t *testing.T) {
repo := NewFSRepository(fstest.MapFS{
"prompts/nested/prompt.yaml": &fstest.MapFile{Data: []byte(`
id: fs-prompt
version: "1.0.0"
inputs:
- name: transcript
required: true
messages:
- role: user
content_file: ./messages/user.tmpl
output:
format: markdown
validation_mode: basic
repair_attempts: 0
`)},
"prompts/nested/messages/user.tmpl": &fstest.MapFile{Data: []byte(`Summarize {{input "transcript"}}.`)},
}, "prompts")
got, err := repo.GetPromptDefinition(context.Background(), "fs-prompt", "")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if got.ID != "fs-prompt" {
t.Fatalf("unexpected prompt id: %q", got.ID)
}
if len(got.Templates) != 1 || !strings.Contains(got.Templates[0].Content, `{{input "transcript"}}`) {
t.Fatalf("expected content_file body to be loaded, got %+v", got.Templates)
}
if got.Templates[0].ContentFile != "prompts/nested/messages/user.tmpl" {
t.Fatalf("unexpected content file path: %q", got.Templates[0].ContentFile)
}
}
func TestFSRepositoryContentFileContainment(t *testing.T) {
t.Run("nested prompt can reference file inside root", func(t *testing.T) {
repo := NewFSRepository(fstest.MapFS{
"prompts/nested/prompt.yaml": &fstest.MapFile{Data: []byte(`
id: fs-contained-prompt
version: "1.0.0"
messages:
- role: user
content_file: ../shared/user.tmpl
output:
format: markdown
validation_mode: basic
repair_attempts: 0
`)},
"prompts/shared/user.tmpl": &fstest.MapFile{Data: []byte(`Inside root.`)},
}, "prompts")
got, err := repo.GetPromptDefinition(context.Background(), "fs-contained-prompt", "")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if len(got.Templates) != 1 || got.Templates[0].Content != "Inside root." {
t.Fatalf("expected contained content file, got %+v", got.Templates)
}
})
tests := []struct {
name string
contentFile string
wantErr string
}{
{name: "parent escape rejected", contentFile: "../outside.tmpl", wantErr: "escapes source root"},
{name: "absolute path rejected", contentFile: "/outside.tmpl", wantErr: "must be relative"},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
repo := NewFSRepository(fstest.MapFS{
"prompts/prompt.yaml": &fstest.MapFile{Data: []byte(`
id: fs-escaped-prompt
version: "1.0.0"
messages:
- role: user
content_file: ` + tc.contentFile + `
output:
format: markdown
validation_mode: basic
repair_attempts: 0
`)},
"outside.tmpl": &fstest.MapFile{Data: []byte(`Outside root.`)},
}, "prompts")
_, err := repo.GetPromptDefinition(context.Background(), "fs-escaped-prompt", "")
if !errors.Is(err, ErrInvalidPromptDefinition) {
t.Fatalf("expected ErrInvalidPromptDefinition, got %v", err)
}
if !strings.Contains(err.Error(), tc.wantErr) {
t.Fatalf("expected error to contain %q, got %v", tc.wantErr, err)
}
})
}
}
func TestFSRepositoryRejectsDuplicatePromptIDs(t *testing.T) {
repo := NewFSRepository(fstest.MapFS{
"one.yaml": &fstest.MapFile{Data: []byte(`
id: duplicate-fs-prompt
version: "1.0.0"
messages:
- role: user
content: First.
output:
format: text
validation_mode: none
repair_attempts: 0
`)},
"nested/two.yaml": &fstest.MapFile{Data: []byte(`
id: duplicate-fs-prompt
version: "1.0.0"
messages:
- role: user
content: Second.
output:
format: text
validation_mode: none
repair_attempts: 0
`)},
}, ".")
_, err := repo.GetPromptDefinition(context.Background(), "duplicate-fs-prompt", "")
if !errors.Is(err, ErrInvalidPromptDefinition) {
t.Fatalf("expected ErrInvalidPromptDefinition, got %v", err)
}
if !strings.Contains(err.Error(), "one.yaml") || !strings.Contains(err.Error(), "nested/two.yaml") {
t.Fatalf("expected duplicate paths in error, got %v", err)
}
}
func TestFSRepositoryRejectsUnknownYAMLFields(t *testing.T) {
repo := NewFSRepository(fstest.MapFS{
"not_named_like_id.yaml": &fstest.MapFile{Data: []byte(`
id: strict-fs-prompt
version: "1.0.0"
unknown: true
messages:
- role: user
content: Invalid.
output:
format: text
validation_mode: none
repair_attempts: 0
`)},
}, ".")
_, err := repo.GetPromptDefinition(context.Background(), "strict-fs-prompt", "")
if !errors.Is(err, ErrInvalidYAML) {
t.Fatalf("expected ErrInvalidYAML, got %v", err)
}
}
func assertCacheControl(t *testing.T, got *domain.CacheControl, wantType domain.CacheControlType, wantTTL string) {
t.Helper()
if got == nil {
t.Fatal("expected cache control, got nil")
}
if got.Type != wantType {
t.Fatalf("unexpected cache control type: got %q want %q", got.Type, wantType)
}
if got.TTL != wantTTL {
t.Fatalf("unexpected cache control ttl: got %q want %q", got.TTL, wantTTL)
}
}
func writePromptTestFile(t *testing.T, path string, content string) { func writePromptTestFile(t *testing.T, path string, content string) {
t.Helper() t.Helper()
if err := os.WriteFile(path, []byte(strings.TrimLeft(content, "\n")), 0o644); err != nil { if err := os.WriteFile(path, []byte(strings.TrimLeft(content, "\n")), 0o644); err != nil {

View File

@@ -0,0 +1,10 @@
id: empty-cache-control-type
version: "1.0.0"
messages:
- role: system
content: "Use cached instructions."
cache_control: {}
output:
format: markdown
validation_mode: basic
repair_attempts: 0

View File

@@ -0,0 +1,12 @@
id: unknown-cache-control-field
version: "1.0.0"
messages:
- role: system
content: "Use cached instructions."
cache_control:
type: ephemeral
unexpected: true
output:
format: markdown
validation_mode: basic
repair_attempts: 0

View File

@@ -0,0 +1,12 @@
id: unsupported-cache-control-ttl
version: "1.0.0"
messages:
- role: system
content: "Use cached instructions."
cache_control:
type: ephemeral
ttl: 5m
output:
format: markdown
validation_mode: basic
repair_attempts: 0

View File

@@ -0,0 +1,11 @@
id: unsupported-cache-control-type
version: "1.0.0"
messages:
- role: system
content: "Use cached instructions."
cache_control:
type: persistent
output:
format: markdown
validation_mode: basic
repair_attempts: 0

View File

@@ -0,0 +1,14 @@
id: valid-cache-control-ttl
version: "1.0.0"
messages:
- role: system
content: "Use cached instructions."
cache_control:
type: ephemeral
ttl: 1h
- role: user
content: "Summarize the input."
output:
format: markdown
validation_mode: basic
repair_attempts: 0

View File

@@ -0,0 +1,13 @@
id: valid-cache-control-without-ttl
version: "1.0.0"
messages:
- role: system
content: "Use cached instructions."
cache_control:
type: ephemeral
- role: user
content: "Summarize the input."
output:
format: markdown
validation_mode: basic
repair_attempts: 0

View File

@@ -0,0 +1,10 @@
id: valid-session-id
version: "1.0.0"
session_id: " {{ .session_id }} "
messages:
- role: user
content: Hello.
output:
format: markdown
validation_mode: basic
repair_attempts: 0

View File

@@ -27,7 +27,9 @@ var (
ErrInvalidRequest = errors.New("invalid run request") ErrInvalidRequest = errors.New("invalid run request")
ErrProfileRequired = errors.New("profile selection is required") ErrProfileRequired = errors.New("profile selection is required")
ErrAPIKeyEnvMissing = errors.New("api_key_env points to an unset environment variable") ErrAPIKeyEnvMissing = errors.New("api_key_env points to an unset environment variable")
ErrProfileLoad = errors.New("failed to load prompt definition") ErrAPIKeyRequired = errors.New("api key is required")
ErrPromptLoad = errors.New("failed to load prompt definition")
ErrProfileLoad = errors.New("failed to load execution profile")
ErrArtifactLoad = errors.New("failed to load artifact") ErrArtifactLoad = errors.New("failed to load artifact")
ErrPromptRender = errors.New("failed to render prompt") ErrPromptRender = errors.New("failed to render prompt")
ErrLLMGenerate = errors.New("failed to generate output") ErrLLMGenerate = errors.New("failed to generate output")
@@ -90,11 +92,15 @@ func (r *Runner) Run(ctx context.Context, req domain.RunRequest) (*domain.RunRes
} }
genResp, err := r.llm.Generate(ctx, domain.GenerateRequest{ genResp, err := r.llm.Generate(ctx, domain.GenerateRequest{
Prompt: domain.RenderedPrompt{Messages: prepared.Messages}, Prompt: domain.RenderedPrompt{SessionID: prepared.SessionID, Messages: prepared.Messages},
Target: prepared.EffectiveModelParams, Target: prepared.EffectiveModelParams,
TargetPresence: prepared.TargetPresence,
StructuredOutput: prepared.StructuredOutput, StructuredOutput: prepared.StructuredOutput,
}) })
if err != nil { if err != nil {
if errors.Is(err, llm.ErrInvalidRequest) {
return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err)
}
return nil, fmt.Errorf("%w: %w", ErrLLMGenerate, err) return nil, fmt.Errorf("%w: %w", ErrLLMGenerate, err)
} }
@@ -167,11 +173,11 @@ func (r *Runner) Prepare(ctx context.Context, req domain.RunRequest) (*domain.Pr
def, err := r.promptDefs.GetPromptDefinition(ctx, req.PromptID, req.PromptVersion) def, err := r.promptDefs.GetPromptDefinition(ctx, req.PromptID, req.PromptVersion)
if err != nil { if err != nil {
return nil, fmt.Errorf("%w: %w", ErrProfileLoad, err) return nil, fmt.Errorf("%w: %w", ErrPromptLoad, err)
} }
promptDefinitionHash, err := hashPromptDefinition(def) promptDefinitionHash, err := hashPromptDefinition(def)
if err != nil { if err != nil {
return nil, fmt.Errorf("%w: failed to hash prompt definition: %v", ErrProfileLoad, err) return nil, fmt.Errorf("%w: failed to hash prompt definition: %v", ErrPromptLoad, err)
} }
selectedProfileID := strings.TrimSpace(req.ProfileID) selectedProfileID := strings.TrimSpace(req.ProfileID)
@@ -187,14 +193,18 @@ func (r *Runner) Prepare(ctx context.Context, req domain.RunRequest) (*domain.Pr
return nil, fmt.Errorf("%w: %w", ErrProfileLoad, err) return nil, fmt.Errorf("%w: %w", ErrProfileLoad, err)
} }
effectiveModel := resolveExecutionTarget(execProfile, req.Execution) effectiveModel, targetPresence, err := resolveExecutionTarget(execProfile, req.Execution)
if err != nil {
return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err)
}
effectiveModel.APIKey = req.APIKey
if strings.TrimSpace(effectiveModel.Endpoint) == "" { if strings.TrimSpace(effectiveModel.Endpoint) == "" {
return nil, fmt.Errorf("%w: execution endpoint is required", ErrInvalidRequest) return nil, fmt.Errorf("%w: execution endpoint is required", ErrInvalidRequest)
} }
if strings.TrimSpace(effectiveModel.Model) == "" { if strings.TrimSpace(effectiveModel.Model) == "" {
return nil, fmt.Errorf("%w: execution model is required", ErrInvalidRequest) return nil, fmt.Errorf("%w: execution model is required", ErrInvalidRequest)
} }
if err := validateAPIKeyEnv(effectiveModel.APIKeyEnv); err != nil { if err := validateAPIKey(effectiveModel.APIKeyEnv, effectiveModel.APIKey, effectiveModel.APIKeyRequired); err != nil {
return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err) return nil, fmt.Errorf("%w: %w", ErrInvalidRequest, err)
} }
@@ -230,9 +240,11 @@ func (r *Runner) Prepare(ctx context.Context, req domain.RunRequest) (*domain.Pr
PromptHash: promptDefinitionHash, PromptHash: promptDefinitionHash,
SelectedProfileID: selectedProfileID, SelectedProfileID: selectedProfileID,
EffectiveModelParams: effectiveModel, EffectiveModelParams: effectiveModel,
TargetPresence: targetPresence,
OutputContract: effectiveContract, OutputContract: effectiveContract,
StructuredOutput: structuredOutput, StructuredOutput: structuredOutput,
InputHashes: inputHashes, InputHashes: inputHashes,
SessionID: renderedPrompt.SessionID,
RenderedPromptHash: hashRenderedPrompt(*renderedPrompt), RenderedPromptHash: hashRenderedPrompt(*renderedPrompt),
Messages: renderedPrompt.Messages, Messages: renderedPrompt.Messages,
StartTime: start, StartTime: start,
@@ -353,28 +365,90 @@ func mergeExecutionTarget(base domain.ExecutionTarget, override domain.Execution
if strings.TrimSpace(override.APIKeyEnv) != "" { if strings.TrimSpace(override.APIKeyEnv) != "" {
out.APIKeyEnv = override.APIKeyEnv out.APIKeyEnv = override.APIKeyEnv
} }
if override.APIKeyRequired {
out.APIKeyRequired = true
}
if len(override.ExtraParams) > 0 { if len(override.ExtraParams) > 0 {
cp := make(map[string]string, len(override.ExtraParams)) out.ExtraParams = copyExtraParams(override.ExtraParams)
for k, v := range override.ExtraParams {
cp[k] = v
}
out.ExtraParams = cp
} }
return out return out
} }
func resolveExecutionTarget(profileValue *domain.ExecutionProfile, override *domain.ExecutionTarget) domain.ExecutionTarget { func mergeExecutionTargetOverride(base domain.ExecutionTarget, override domain.ExecutionTargetOverride) (domain.ExecutionTarget, domain.ExecutionTargetPresence, error) {
out := base
var presence domain.ExecutionTargetPresence
if override.Endpoint != "" {
out.Endpoint = override.Endpoint
}
if override.Model != "" {
out.Model = override.Model
}
if override.Temperature != nil {
if *override.Temperature < 0 || *override.Temperature > 2 {
return domain.ExecutionTarget{}, domain.ExecutionTargetPresence{}, errors.New("temperature must be between 0 and 2")
}
out.Temperature = *override.Temperature
presence.Temperature = true
}
if override.MaxTokens != nil {
if *override.MaxTokens < 0 {
return domain.ExecutionTarget{}, domain.ExecutionTargetPresence{}, errors.New("max_tokens must be greater than or equal to 0")
}
out.MaxTokens = *override.MaxTokens
presence.MaxTokens = true
}
if override.TopP != nil {
if *override.TopP < 0 || *override.TopP > 1 {
return domain.ExecutionTarget{}, domain.ExecutionTargetPresence{}, errors.New("top_p must be between 0 and 1")
}
out.TopP = *override.TopP
presence.TopP = true
}
if override.TimeoutSeconds != nil {
if *override.TimeoutSeconds < 0 {
return domain.ExecutionTarget{}, domain.ExecutionTargetPresence{}, errors.New("timeout_seconds must be greater than or equal to 0")
}
out.TimeoutSeconds = *override.TimeoutSeconds
presence.TimeoutSeconds = true
}
if strings.TrimSpace(override.ServiceTier) != "" {
out.ServiceTier = override.ServiceTier
}
if strings.TrimSpace(override.ReasoningEffort) != "" {
out.ReasoningEffort = override.ReasoningEffort
}
if strings.TrimSpace(override.APIKeyEnv) != "" {
out.APIKeyEnv = override.APIKeyEnv
}
if len(override.ExtraParams) > 0 {
out.ExtraParams = copyExtraParams(override.ExtraParams)
}
return out, presence, nil
}
func resolveExecutionTarget(profileValue *domain.ExecutionProfile, override *domain.ExecutionTargetOverride) (domain.ExecutionTarget, domain.ExecutionTargetPresence, error) {
out := defaults.ExecutionTargetDefault() out := defaults.ExecutionTargetDefault()
out = mergeExecutionTarget(out, executionProfileToTarget(profileValue)) out = mergeExecutionTarget(out, executionProfileToTarget(profileValue))
var presence domain.ExecutionTargetPresence
if override != nil { if override != nil {
out = mergeExecutionTarget(out, *override) var err error
out, presence, err = mergeExecutionTargetOverride(out, *override)
if err != nil {
return domain.ExecutionTarget{}, domain.ExecutionTargetPresence{}, err
}
} }
return out return out, presence, nil
} }
func validateAPIKeyEnv(apiKeyEnv string) error { func validateAPIKey(apiKeyEnv string, apiKey string, apiKeyRequired bool) error {
if strings.TrimSpace(apiKey) != "" {
return nil
}
envName := strings.TrimSpace(apiKeyEnv) envName := strings.TrimSpace(apiKeyEnv)
if envName == "" { if envName == "" {
if apiKeyRequired {
return ErrAPIKeyRequired
}
return nil return nil
} }
if strings.TrimSpace(os.Getenv(envName)) == "" { if strings.TrimSpace(os.Getenv(envName)) == "" {
@@ -387,13 +461,6 @@ func executionProfileToTarget(p *domain.ExecutionProfile) domain.ExecutionTarget
if p == nil { if p == nil {
return domain.ExecutionTarget{} return domain.ExecutionTarget{}
} }
cp := map[string]string(nil)
if len(p.ExtraParams) > 0 {
cp = make(map[string]string, len(p.ExtraParams))
for k, v := range p.ExtraParams {
cp[k] = v
}
}
return domain.ExecutionTarget{ return domain.ExecutionTarget{
Endpoint: p.Endpoint, Endpoint: p.Endpoint,
Model: p.Model, Model: p.Model,
@@ -404,10 +471,22 @@ func executionProfileToTarget(p *domain.ExecutionProfile) domain.ExecutionTarget
ServiceTier: p.ServiceTier, ServiceTier: p.ServiceTier,
ReasoningEffort: p.ReasoningEffort, ReasoningEffort: p.ReasoningEffort,
APIKeyEnv: p.APIKeyEnv, APIKeyEnv: p.APIKeyEnv,
ExtraParams: cp, APIKeyRequired: p.APIKeyRequired,
ExtraParams: copyExtraParams(p.ExtraParams),
} }
} }
func copyExtraParams(src map[string]any) map[string]any {
if len(src) == 0 {
return nil
}
cp := make(map[string]any, len(src))
for k, v := range src {
cp[k] = v
}
return cp
}
func resolveOutputContract(def *domain.PromptDefinition, override *domain.OutputContract) domain.OutputContract { func resolveOutputContract(def *domain.PromptDefinition, override *domain.OutputContract) domain.OutputContract {
contract := def.Validation contract := def.Validation
if contract.Format == "" { if contract.Format == "" {
@@ -424,10 +503,23 @@ func resolveOutputContract(def *domain.PromptDefinition, override *domain.Output
func hashRenderedPrompt(p domain.RenderedPrompt) string { func hashRenderedPrompt(p domain.RenderedPrompt) string {
var b strings.Builder var b strings.Builder
if p.SessionID != "" {
b.WriteString("session_id=")
b.WriteString(p.SessionID)
b.WriteString("\n---\n")
}
for _, msg := range p.Messages { for _, msg := range p.Messages {
b.WriteString(msg.Role) b.WriteString(msg.Role)
b.WriteByte('\n') b.WriteByte('\n')
b.WriteString(msg.Content) b.WriteString(msg.Content)
if msg.CacheControl != nil {
b.WriteString("\ncache_control.type=")
b.WriteString(string(msg.CacheControl.Type))
if msg.CacheControl.TTL != "" {
b.WriteString("\ncache_control.ttl=")
b.WriteString(msg.CacheControl.TTL)
}
}
b.WriteString("\n---\n") b.WriteString("\n---\n")
} }
h := sha256.Sum256([]byte(b.String())) h := sha256.Sum256([]byte(b.String()))

View File

@@ -14,6 +14,7 @@ import (
"gitea.maximumdirect.net/eric/scriptorium/internal/defaults" "gitea.maximumdirect.net/eric/scriptorium/internal/defaults"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
"gitea.maximumdirect.net/eric/scriptorium/internal/llm"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile" "gitea.maximumdirect.net/eric/scriptorium/internal/profile"
"gitea.maximumdirect.net/eric/scriptorium/internal/prompt" "gitea.maximumdirect.net/eric/scriptorium/internal/prompt"
"gitea.maximumdirect.net/eric/scriptorium/internal/promptdef" "gitea.maximumdirect.net/eric/scriptorium/internal/promptdef"
@@ -160,7 +161,7 @@ func TestRunnerPrepareWithExplicitProfileSelection(t *testing.T) {
"a://t": {Body: []byte("transcript"), Hash: hashString("transcript")}, "a://t": {Body: []byte("transcript"), Hash: hashString("transcript")},
"a://g": {Body: []byte("glossary"), Hash: hashString("glossary")}, "a://g": {Body: []byte("glossary"), Hash: hashString("glossary")},
}} }}
renderer := &fakeRenderer{rendered: &domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "system", Content: "sys"}, {Role: "user", Content: "usr"}}}} renderer := &fakeRenderer{rendered: &domain.RenderedPrompt{SessionID: "session-123", Messages: []domain.RenderedMessage{{Role: "system", Content: "sys"}, {Role: "user", Content: "usr"}}}}
llmClient := &fakeLLM{forbid: true} llmClient := &fakeLLM{forbid: true}
runner := NewRunner(promptRepo, execRepo, reader, renderer, llmClient, nil) runner := NewRunner(promptRepo, execRepo, reader, renderer, llmClient, nil)
@@ -172,7 +173,7 @@ func TestRunnerPrepareWithExplicitProfileSelection(t *testing.T) {
"transcript": {Type: domain.ArtifactRefFile, URI: "a://t"}, "transcript": {Type: domain.ArtifactRefFile, URI: "a://t"},
"glossary": {Type: domain.ArtifactRefFile, URI: "a://g"}, "glossary": {Type: domain.ArtifactRefFile, URI: "a://g"},
}, },
Execution: &domain.ExecutionTarget{Endpoint: "http://override/v1", Model: "m", Temperature: 0.3, TimeoutSeconds: 90}, Execution: &domain.ExecutionTargetOverride{Endpoint: "http://override/v1", Model: "m", Temperature: float64Ptr(0.3), TimeoutSeconds: intPtr(90)},
}) })
if err != nil { if err != nil {
t.Fatalf("expected no error, got %v", err) t.Fatalf("expected no error, got %v", err)
@@ -198,6 +199,9 @@ func TestRunnerPrepareWithExplicitProfileSelection(t *testing.T) {
if len(prepared.Messages) != 2 { if len(prepared.Messages) != 2 {
t.Fatalf("expected two messages, got %d", len(prepared.Messages)) t.Fatalf("expected two messages, got %d", len(prepared.Messages))
} }
if prepared.SessionID != "session-123" {
t.Fatalf("expected prepared session id, got %q", prepared.SessionID)
}
if llmClient.calls != 0 { if llmClient.calls != 0 {
t.Fatalf("prepare should not call llm, calls=%d", llmClient.calls) t.Fatalf("prepare should not call llm, calls=%d", llmClient.calls)
} }
@@ -249,6 +253,17 @@ func TestRunnerPrepareSelectedProfileDoesNotExistFails(t *testing.T) {
} }
} }
func TestRunnerPreparePromptLoadFailure(t *testing.T) {
runner := NewRunner(&fakePromptRepo{err: errors.New("boom")}, &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}}, defaultArtifactReader(), defaultRenderer(), &fakeLLM{}, nil)
_, err := runner.Prepare(context.Background(), domain.RunRequest{PromptID: "p"})
if !errors.Is(err, ErrPromptLoad) {
t.Fatalf("expected ErrPromptLoad, got %v", err)
}
if errors.Is(err, ErrProfileLoad) {
t.Fatalf("did not expect ErrProfileLoad, got %v", err)
}
}
func TestRunnerPrepareRuntimeOverrideBeatsSelectedProfileValue(t *testing.T) { func TestRunnerPrepareRuntimeOverrideBeatsSelectedProfileValue(t *testing.T) {
promptRepo := &fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)} promptRepo := &fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)}
execRepo := &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{ execRepo := &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{
@@ -269,11 +284,11 @@ func TestRunnerPrepareRuntimeOverrideBeatsSelectedProfileValue(t *testing.T) {
PromptID: "p", PromptID: "p",
ProfileID: "exec", ProfileID: "exec",
Inputs: singleInputRef(), Inputs: singleInputRef(),
Execution: &domain.ExecutionTarget{ Execution: &domain.ExecutionTargetOverride{
Endpoint: "http://override/v1", Endpoint: "http://override/v1",
Model: "override-model", Model: "override-model",
Temperature: 0.7, Temperature: float64Ptr(0.7),
TimeoutSeconds: 30, TimeoutSeconds: intPtr(30),
ServiceTier: "flex", ServiceTier: "flex",
}, },
}) })
@@ -291,6 +306,143 @@ func TestRunnerPrepareRuntimeOverrideBeatsSelectedProfileValue(t *testing.T) {
} }
} }
func TestRunnerPrepareRequestNumericOverridePresence(t *testing.T) {
tests := []struct {
name string
override *domain.ExecutionTargetOverride
wantTemperature float64
wantMaxTokens int
wantTopP float64
wantTimeoutSecs int
wantPresence domain.ExecutionTargetPresence
}{
{
name: "omitted preserves profile values",
override: &domain.ExecutionTargetOverride{},
wantTemperature: 0.7,
wantMaxTokens: 321,
wantTopP: 0.8,
wantTimeoutSecs: 45,
},
{
name: "explicit zero temperature",
override: &domain.ExecutionTargetOverride{Temperature: float64Ptr(0)},
wantTemperature: 0,
wantMaxTokens: 321,
wantTopP: 0.8,
wantTimeoutSecs: 45,
wantPresence: domain.ExecutionTargetPresence{Temperature: true},
},
{
name: "explicit zero max tokens",
override: &domain.ExecutionTargetOverride{MaxTokens: intPtr(0)},
wantTemperature: 0.7,
wantMaxTokens: 0,
wantTopP: 0.8,
wantTimeoutSecs: 45,
wantPresence: domain.ExecutionTargetPresence{MaxTokens: true},
},
{
name: "explicit zero top p",
override: &domain.ExecutionTargetOverride{TopP: float64Ptr(0)},
wantTemperature: 0.7,
wantMaxTokens: 321,
wantTopP: 0,
wantTimeoutSecs: 45,
wantPresence: domain.ExecutionTargetPresence{TopP: true},
},
{
name: "explicit zero timeout",
override: &domain.ExecutionTargetOverride{TimeoutSeconds: intPtr(0)},
wantTemperature: 0.7,
wantMaxTokens: 321,
wantTopP: 0.8,
wantTimeoutSecs: 0,
wantPresence: domain.ExecutionTargetPresence{TimeoutSeconds: true},
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
runner := NewRunner(
&fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)},
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{
"exec": {
ID: "exec",
Endpoint: "http://profile/v1",
Model: "profile-model",
Temperature: 0.7,
MaxTokens: 321,
TopP: 0.8,
TimeoutSeconds: 45,
},
}},
defaultArtifactReader(),
defaultRenderer(),
&fakeLLM{forbid: true},
nil,
)
prepared, err := runner.Prepare(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
Inputs: singleInputRef(),
Execution: tc.override,
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
got := prepared.EffectiveModelParams
if got.Temperature != tc.wantTemperature ||
got.MaxTokens != tc.wantMaxTokens ||
got.TopP != tc.wantTopP ||
got.TimeoutSeconds != tc.wantTimeoutSecs {
t.Fatalf("unexpected effective numeric settings: %+v", got)
}
if prepared.TargetPresence != tc.wantPresence {
t.Fatalf("unexpected target presence: got %+v want %+v", prepared.TargetPresence, tc.wantPresence)
}
})
}
}
func TestRunnerPrepareInvalidRequestNumericOverridesFail(t *testing.T) {
tests := []struct {
name string
override *domain.ExecutionTargetOverride
}{
{name: "temperature below range", override: &domain.ExecutionTargetOverride{Temperature: float64Ptr(-0.1)}},
{name: "temperature above range", override: &domain.ExecutionTargetOverride{Temperature: float64Ptr(2.1)}},
{name: "max tokens below range", override: &domain.ExecutionTargetOverride{MaxTokens: intPtr(-1)}},
{name: "top p below range", override: &domain.ExecutionTargetOverride{TopP: float64Ptr(-0.1)}},
{name: "top p above range", override: &domain.ExecutionTargetOverride{TopP: float64Ptr(1.1)}},
{name: "timeout below range", override: &domain.ExecutionTargetOverride{TimeoutSeconds: intPtr(-1)}},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
runner := NewRunner(
&fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)},
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}},
defaultArtifactReader(),
defaultRenderer(),
&fakeLLM{forbid: true},
nil,
)
_, err := runner.Prepare(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
Inputs: singleInputRef(),
Execution: tc.override,
})
if !errors.Is(err, ErrInvalidRequest) {
t.Fatalf("expected ErrInvalidRequest, got %v", err)
}
})
}
}
func TestRunnerPrepareSelectedProfileBeatsBuiltInDefault(t *testing.T) { func TestRunnerPrepareSelectedProfileBeatsBuiltInDefault(t *testing.T) {
promptRepo := &fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)} promptRepo := &fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)}
execRepo := &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{ execRepo := &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{
@@ -577,6 +729,100 @@ func TestDeriveStructuredSchemaName(t *testing.T) {
} }
} }
func TestHashRenderedPromptIncludesCacheControlWhenPresent(t *testing.T) {
uncached := domain.RenderedPrompt{Messages: []domain.RenderedMessage{
{Role: "system", Content: "sys"},
{Role: "user", Content: "usr"},
}}
wantLegacyHash := hashString("system\nsys\n---\nuser\nusr\n---\n")
if got := hashRenderedPrompt(uncached); got != wantLegacyHash {
t.Fatalf("expected no-cache hash to preserve legacy input, got %q want %q", got, wantLegacyHash)
}
withCache := domain.RenderedPrompt{Messages: []domain.RenderedMessage{
{
Role: "system",
Content: "sys",
CacheControl: &domain.CacheControl{
Type: domain.CacheControlEphemeral,
TTL: "1h",
},
},
{Role: "user", Content: "usr"},
}}
alsoWithCache := domain.RenderedPrompt{Messages: []domain.RenderedMessage{
{
Role: "system",
Content: "sys",
CacheControl: &domain.CacheControl{
Type: domain.CacheControlEphemeral,
TTL: "1h",
},
},
{Role: "user", Content: "usr"},
}}
withoutTTL := domain.RenderedPrompt{Messages: []domain.RenderedMessage{
{
Role: "system",
Content: "sys",
CacheControl: &domain.CacheControl{
Type: domain.CacheControlEphemeral,
},
},
{Role: "user", Content: "usr"},
}}
cachedHash := hashRenderedPrompt(withCache)
if cachedHash == hashRenderedPrompt(uncached) {
t.Fatal("expected cache control to change rendered prompt hash")
}
if cachedHash != hashRenderedPrompt(alsoWithCache) {
t.Fatal("expected identical cache control metadata to produce stable hash")
}
if cachedHash == hashRenderedPrompt(withoutTTL) {
t.Fatal("expected ttl changes to affect rendered prompt hash")
}
}
func TestHashRenderedPromptIncludesSessionIDWhenPresent(t *testing.T) {
withoutSession := domain.RenderedPrompt{Messages: []domain.RenderedMessage{
{Role: "system", Content: "sys"},
{Role: "user", Content: "usr"},
}}
withSession := domain.RenderedPrompt{
SessionID: "session-123",
Messages: []domain.RenderedMessage{
{Role: "system", Content: "sys"},
{Role: "user", Content: "usr"},
},
}
alsoWithSession := domain.RenderedPrompt{
SessionID: "session-123",
Messages: []domain.RenderedMessage{
{Role: "system", Content: "sys"},
{Role: "user", Content: "usr"},
},
}
otherSession := domain.RenderedPrompt{
SessionID: "session-456",
Messages: []domain.RenderedMessage{
{Role: "system", Content: "sys"},
{Role: "user", Content: "usr"},
},
}
sessionHash := hashRenderedPrompt(withSession)
if sessionHash == hashRenderedPrompt(withoutSession) {
t.Fatal("expected session_id to change rendered prompt hash")
}
if sessionHash != hashRenderedPrompt(alsoWithSession) {
t.Fatal("expected identical session_id to produce stable hash")
}
if sessionHash == hashRenderedPrompt(otherSession) {
t.Fatal("expected session_id value changes to affect rendered prompt hash")
}
}
func TestRunnerRunSuccessful(t *testing.T) { func TestRunnerRunSuccessful(t *testing.T) {
promptRepo := &fakePromptRepo{def: promptDef(domain.FormatMarkdown, domain.ValidationBasic, 0)} promptRepo := &fakePromptRepo{def: promptDef(domain.FormatMarkdown, domain.ValidationBasic, 0)}
execRepo := &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}} execRepo := &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}}
@@ -584,7 +830,7 @@ func TestRunnerRunSuccessful(t *testing.T) {
"a://t": {Body: []byte("transcript"), Hash: hashString("transcript")}, "a://t": {Body: []byte("transcript"), Hash: hashString("transcript")},
"a://g": {Body: []byte("glossary"), Hash: hashString("glossary")}, "a://g": {Body: []byte("glossary"), Hash: hashString("glossary")},
}} }}
renderer := &fakeRenderer{rendered: &domain.RenderedPrompt{Messages: []domain.RenderedMessage{{Role: "system", Content: "sys"}, {Role: "user", Content: "usr"}}}} renderer := &fakeRenderer{rendered: &domain.RenderedPrompt{SessionID: "session-123", Messages: []domain.RenderedMessage{{Role: "system", Content: "sys"}, {Role: "user", Content: "usr"}}}}
llmClient := &fakeLLM{resp: &domain.GenerateResponse{Content: "# recap", Usage: domain.TokenUsage{TotalTokens: 7}}} llmClient := &fakeLLM{resp: &domain.GenerateResponse{Content: "# recap", Usage: domain.TokenUsage{TotalTokens: 7}}}
runner := NewRunner(promptRepo, execRepo, reader, renderer, llmClient, nil) runner := NewRunner(promptRepo, execRepo, reader, renderer, llmClient, nil)
@@ -596,7 +842,7 @@ func TestRunnerRunSuccessful(t *testing.T) {
"transcript": {Type: domain.ArtifactRefFile, URI: "a://t"}, "transcript": {Type: domain.ArtifactRefFile, URI: "a://t"},
"glossary": {Type: domain.ArtifactRefFile, URI: "a://g"}, "glossary": {Type: domain.ArtifactRefFile, URI: "a://g"},
}, },
Execution: &domain.ExecutionTarget{Endpoint: "http://override/v1", Model: "m", Temperature: 0.3, TimeoutSeconds: 90}, Execution: &domain.ExecutionTargetOverride{Endpoint: "http://override/v1", Model: "m", Temperature: float64Ptr(0.3), TimeoutSeconds: intPtr(90)},
}) })
if err != nil { if err != nil {
t.Fatalf("expected no error, got %v", err) t.Fatalf("expected no error, got %v", err)
@@ -631,6 +877,48 @@ func TestRunnerRunSuccessful(t *testing.T) {
if llmClient.lastReq.Target.TimeoutSeconds != 90 { if llmClient.lastReq.Target.TimeoutSeconds != 90 {
t.Fatalf("expected timeout propagation, got %d", llmClient.lastReq.Target.TimeoutSeconds) t.Fatalf("expected timeout propagation, got %d", llmClient.lastReq.Target.TimeoutSeconds)
} }
if !llmClient.lastReq.TargetPresence.Temperature || !llmClient.lastReq.TargetPresence.TimeoutSeconds {
t.Fatalf("expected numeric override presence to be sent to llm, got %+v", llmClient.lastReq.TargetPresence)
}
if llmClient.lastReq.Prompt.SessionID != "session-123" {
t.Fatalf("expected session id to be sent to llm, got %q", llmClient.lastReq.Prompt.SessionID)
}
}
func TestRunnerRunPassesExtraParamsToGenerateRequestTarget(t *testing.T) {
extraParams := map[string]any{
"string_value": "enabled",
"number_value": 42,
"boolean_value": true,
"object_value": map[string]any{"nested": "value"},
"array_value": []any{"first", 3, false},
}
promptRepo := &fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)}
execRepo := &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{
"exec": {
ID: "exec",
Endpoint: "http://profile/v1",
Model: "profile-model",
ExtraParams: extraParams,
},
}}
llmClient := &fakeLLM{resp: &domain.GenerateResponse{Content: "ok"}}
runner := NewRunner(promptRepo, execRepo, defaultArtifactReader(), defaultRenderer(), llmClient, nil)
res, err := runner.Run(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
Inputs: singleInputRef(),
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if !reflect.DeepEqual(res.EffectiveModelParams.ExtraParams, extraParams) {
t.Fatalf("expected run result extra_params to match profile values, got %#v", res.EffectiveModelParams.ExtraParams)
}
if !reflect.DeepEqual(llmClient.lastReq.Target.ExtraParams, extraParams) {
t.Fatalf("expected generate request extra_params to match profile values, got %#v", llmClient.lastReq.Target.ExtraParams)
}
} }
func TestRunnerRunAndPrepareResolveSameProfileAndEffectiveSettings(t *testing.T) { func TestRunnerRunAndPrepareResolveSameProfileAndEffectiveSettings(t *testing.T) {
@@ -649,7 +937,7 @@ func TestRunnerRunAndPrepareResolveSameProfileAndEffectiveSettings(t *testing.T)
Inputs: map[string]domain.ArtifactRef{ Inputs: map[string]domain.ArtifactRef{
"transcript": {Type: domain.ArtifactRefFile, URI: "a://t"}, "transcript": {Type: domain.ArtifactRefFile, URI: "a://t"},
}, },
Execution: &domain.ExecutionTarget{Endpoint: "http://override/v1", Model: "m", Temperature: 0.3, TimeoutSeconds: 90}, Execution: &domain.ExecutionTargetOverride{Endpoint: "http://override/v1", Model: "m", Temperature: float64Ptr(0.3), TimeoutSeconds: intPtr(90)},
} }
prepared, err := runner.Prepare(context.Background(), req) prepared, err := runner.Prepare(context.Background(), req)
@@ -777,11 +1065,11 @@ func TestRunnerRunExplicitRuntimeOverrideBeatsSelectedProfileValue(t *testing.T)
PromptID: "p", PromptID: "p",
ProfileID: "exec", ProfileID: "exec",
Inputs: singleInputRef(), Inputs: singleInputRef(),
Execution: &domain.ExecutionTarget{ Execution: &domain.ExecutionTargetOverride{
Endpoint: "http://override/v1", Endpoint: "http://override/v1",
Model: "override-model", Model: "override-model",
Temperature: 0.7, Temperature: float64Ptr(0.7),
TimeoutSeconds: 30, TimeoutSeconds: intPtr(30),
ServiceTier: "flex", ServiceTier: "flex",
}, },
}) })
@@ -902,6 +1190,75 @@ func TestRunnerRunAPIKeyEnvMissingEnvironmentValueFailsClearly(t *testing.T) {
} }
} }
func TestRunnerRunDirectAPIKeyBypassesMissingEnvAndReachesLLM(t *testing.T) {
const directKey = "direct-runner-key"
promptRepo := &fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)}
execRepo := &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{
"exec": {ID: "exec", Endpoint: "http://profile/v1", Model: "profile-model", APIKeyEnv: "SCRIPTORIUM_MISSING_KEY"},
}}
llmClient := &fakeLLM{resp: &domain.GenerateResponse{Content: "ok"}}
runner := NewRunner(promptRepo, execRepo, defaultArtifactReader(), defaultRenderer(), llmClient, nil)
_, err := runner.Run(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
APIKey: directKey,
Inputs: singleInputRef(),
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if llmClient.lastReq.Target.APIKey != directKey {
t.Fatalf("expected direct API key to reach LLM request")
}
if llmClient.lastReq.Target.APIKeyEnv != "SCRIPTORIUM_MISSING_KEY" {
t.Fatalf("expected api_key_env name to remain on target, got %q", llmClient.lastReq.Target.APIKeyEnv)
}
}
func TestRunnerPrepareAPIKeyRequiredFailsWithoutDirectKey(t *testing.T) {
promptRepo := &fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)}
execRepo := &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{
"exec": {ID: "exec", Endpoint: "http://profile/v1", Model: "profile-model", APIKeyRequired: true},
}}
runner := NewRunner(promptRepo, execRepo, defaultArtifactReader(), defaultRenderer(), &fakeLLM{resp: &domain.GenerateResponse{Content: "ok"}}, nil)
_, err := runner.Prepare(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
Inputs: singleInputRef(),
})
if !errors.Is(err, ErrAPIKeyRequired) {
t.Fatalf("expected ErrAPIKeyRequired, got %v", err)
}
}
func TestRunnerRunAPIKeyRequiredSucceedsWithDirectKey(t *testing.T) {
const directKey = "direct-required-key"
promptRepo := &fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)}
execRepo := &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{
"exec": {ID: "exec", Endpoint: "http://profile/v1", Model: "profile-model", APIKeyRequired: true},
}}
llmClient := &fakeLLM{resp: &domain.GenerateResponse{Content: "ok"}}
runner := NewRunner(promptRepo, execRepo, defaultArtifactReader(), defaultRenderer(), llmClient, nil)
_, err := runner.Run(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
APIKey: directKey,
Inputs: singleInputRef(),
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if llmClient.lastReq.Target.APIKey != directKey {
t.Fatalf("expected direct API key to reach LLM request")
}
if !llmClient.lastReq.Target.APIKeyRequired {
t.Fatalf("expected APIKeyRequired to be carried to target")
}
}
func TestRunnerRunRuntimeAPIKeyEnvOverrideWorks(t *testing.T) { func TestRunnerRunRuntimeAPIKeyEnvOverrideWorks(t *testing.T) {
const envName = "SCRIPTORIUM_RUNTIME_API_KEY" const envName = "SCRIPTORIUM_RUNTIME_API_KEY"
t.Setenv(envName, "runtime-secret") t.Setenv(envName, "runtime-secret")
@@ -916,7 +1273,7 @@ func TestRunnerRunRuntimeAPIKeyEnvOverrideWorks(t *testing.T) {
PromptID: "p", PromptID: "p",
ProfileID: "exec", ProfileID: "exec",
Inputs: singleInputRef(), Inputs: singleInputRef(),
Execution: &domain.ExecutionTarget{APIKeyEnv: envName}, Execution: &domain.ExecutionTargetOverride{APIKeyEnv: envName},
}) })
if err != nil { if err != nil {
t.Fatalf("expected no error, got %v", err) t.Fatalf("expected no error, got %v", err)
@@ -941,7 +1298,7 @@ func TestRunnerRunRuntimeAPIKeyEnvOverrideBeatsProfile(t *testing.T) {
PromptID: "p", PromptID: "p",
ProfileID: "exec", ProfileID: "exec",
Inputs: singleInputRef(), Inputs: singleInputRef(),
Execution: &domain.ExecutionTarget{APIKeyEnv: runtimeEnv}, Execution: &domain.ExecutionTargetOverride{APIKeyEnv: runtimeEnv},
}) })
if err != nil { if err != nil {
t.Fatalf("expected no error, got %v", err) t.Fatalf("expected no error, got %v", err)
@@ -977,8 +1334,11 @@ func TestRunnerRunAPIKeyValueNotPresentInMetadata(t *testing.T) {
func TestRunnerRunPromptLoadFailure(t *testing.T) { func TestRunnerRunPromptLoadFailure(t *testing.T) {
runner := NewRunner(&fakePromptRepo{err: errors.New("boom")}, &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}}, defaultArtifactReader(), defaultRenderer(), &fakeLLM{}, nil) runner := NewRunner(&fakePromptRepo{err: errors.New("boom")}, &fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}}, defaultArtifactReader(), defaultRenderer(), &fakeLLM{}, nil)
_, err := runner.Run(context.Background(), domain.RunRequest{PromptID: "p"}) _, err := runner.Run(context.Background(), domain.RunRequest{PromptID: "p"})
if !errors.Is(err, ErrProfileLoad) { if !errors.Is(err, ErrPromptLoad) {
t.Fatalf("expected ErrProfileLoad, got %v", err) t.Fatalf("expected ErrPromptLoad, got %v", err)
}
if errors.Is(err, ErrProfileLoad) {
t.Fatalf("did not expect ErrProfileLoad, got %v", err)
} }
} }
@@ -1040,6 +1400,28 @@ func TestRunnerRunLLMFailure(t *testing.T) {
} }
} }
func TestRunnerRunLLMInvalidRequestMapsToUsecaseInvalidRequest(t *testing.T) {
runner := NewRunner(
&fakePromptRepo{def: promptDef(domain.FormatText, domain.ValidationNone, 0)},
&fakeExecutionProfileRepo{profiles: map[string]*domain.ExecutionProfile{"exec": defaultExecutionProfile()}},
defaultArtifactReader(),
defaultRenderer(),
&fakeLLM{err: llm.ErrInvalidRequest},
nil,
)
_, err := runner.Run(context.Background(), domain.RunRequest{
PromptID: "p",
ProfileID: "exec",
Inputs: singleInputRef(),
})
if !errors.Is(err, ErrInvalidRequest) {
t.Fatalf("expected ErrInvalidRequest, got %v", err)
}
if errors.Is(err, ErrLLMGenerate) {
t.Fatalf("did not expect ErrLLMGenerate, got %v", err)
}
}
func TestRunnerRunValidationStillWorks(t *testing.T) { func TestRunnerRunValidationStillWorks(t *testing.T) {
validator := &fakeValidator{result: domain.ValidationResult{Status: domain.ValidationFailed, Mode: domain.ValidationBasic, Errors: []string{"bad"}, IsValid: false}} validator := &fakeValidator{result: domain.ValidationResult{Status: domain.ValidationFailed, Mode: domain.ValidationBasic, Errors: []string{"bad"}, IsValid: false}}
runner := NewRunner( runner := NewRunner(
@@ -1082,7 +1464,7 @@ func TestRunnerRunStructuredRepairRemainsBoundedAndUsesEffectiveModelSettings(t
PromptID: "p", PromptID: "p",
ProfileID: "exec", ProfileID: "exec",
Inputs: singleInputRef(), Inputs: singleInputRef(),
Execution: &domain.ExecutionTarget{Endpoint: "http://override/v1", Model: "override-model", TimeoutSeconds: 22}, Execution: &domain.ExecutionTargetOverride{Endpoint: "http://override/v1", Model: "override-model", TimeoutSeconds: intPtr(22)},
}) })
if err != nil { if err != nil {
t.Fatalf("expected no error, got %v", err) t.Fatalf("expected no error, got %v", err)
@@ -1169,7 +1551,8 @@ func TestExecutionProfileToTargetPopulatesAllFieldsAndCopiesExtraParams(t *testi
ServiceTier: "priority", ServiceTier: "priority",
ReasoningEffort: "medium", ReasoningEffort: "medium",
APIKeyEnv: "SCRIPTORIUM_API_KEY", APIKeyEnv: "SCRIPTORIUM_API_KEY",
ExtraParams: map[string]string{ APIKeyRequired: true,
ExtraParams: map[string]any{
"provider_option": "on", "provider_option": "on",
}, },
} }
@@ -1183,7 +1566,8 @@ func TestExecutionProfileToTargetPopulatesAllFieldsAndCopiesExtraParams(t *testi
target.TimeoutSeconds != src.TimeoutSeconds || target.TimeoutSeconds != src.TimeoutSeconds ||
target.ServiceTier != src.ServiceTier || target.ServiceTier != src.ServiceTier ||
target.ReasoningEffort != src.ReasoningEffort || target.ReasoningEffort != src.ReasoningEffort ||
target.APIKeyEnv != src.APIKeyEnv { target.APIKeyEnv != src.APIKeyEnv ||
target.APIKeyRequired != src.APIKeyRequired {
t.Fatalf("expected all profile fields to populate target, got %+v", target) t.Fatalf("expected all profile fields to populate target, got %+v", target)
} }
if !reflect.DeepEqual(target.ExtraParams, src.ExtraParams) { if !reflect.DeepEqual(target.ExtraParams, src.ExtraParams) {
@@ -1208,12 +1592,19 @@ func TestResolveExecutionTargetProfileValuesPopulateAllSupportedFields(t *testin
ServiceTier: "priority", ServiceTier: "priority",
ReasoningEffort: "low", ReasoningEffort: "low",
APIKeyEnv: "PROFILE_KEY", APIKeyEnv: "PROFILE_KEY",
ExtraParams: map[string]string{ APIKeyRequired: true,
ExtraParams: map[string]any{
"profile_option": "enabled", "profile_option": "enabled",
}, },
} }
target := resolveExecutionTarget(profileValue, nil) target, presence, err := resolveExecutionTarget(profileValue, nil)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if presence != (domain.ExecutionTargetPresence{}) {
t.Fatalf("expected no request override presence, got %+v", presence)
}
if target.Endpoint != profileValue.Endpoint || if target.Endpoint != profileValue.Endpoint ||
target.Model != profileValue.Model || target.Model != profileValue.Model ||
target.Temperature != profileValue.Temperature || target.Temperature != profileValue.Temperature ||
@@ -1222,7 +1613,8 @@ func TestResolveExecutionTargetProfileValuesPopulateAllSupportedFields(t *testin
target.TimeoutSeconds != profileValue.TimeoutSeconds || target.TimeoutSeconds != profileValue.TimeoutSeconds ||
target.ServiceTier != profileValue.ServiceTier || target.ServiceTier != profileValue.ServiceTier ||
target.ReasoningEffort != profileValue.ReasoningEffort || target.ReasoningEffort != profileValue.ReasoningEffort ||
target.APIKeyEnv != profileValue.APIKeyEnv { target.APIKeyEnv != profileValue.APIKeyEnv ||
target.APIKeyRequired != profileValue.APIKeyRequired {
t.Fatalf("expected profile values to populate target, got %+v", target) t.Fatalf("expected profile values to populate target, got %+v", target)
} }
if !reflect.DeepEqual(target.ExtraParams, profileValue.ExtraParams) { if !reflect.DeepEqual(target.ExtraParams, profileValue.ExtraParams) {
@@ -1242,32 +1634,38 @@ func TestResolveExecutionTargetRuntimeOverridesBeatProfileForAllOverrideableFiel
ServiceTier: "priority", ServiceTier: "priority",
ReasoningEffort: "medium", ReasoningEffort: "medium",
APIKeyEnv: "PROFILE_KEY", APIKeyEnv: "PROFILE_KEY",
ExtraParams: map[string]string{ ExtraParams: map[string]any{
"profile_only": "yes", "profile_only": "yes",
}, },
} }
override := &domain.ExecutionTarget{ override := &domain.ExecutionTargetOverride{
Endpoint: "http://override/v1", Endpoint: "http://override/v1",
Model: "override-model", Model: "override-model",
Temperature: 0.9, Temperature: float64Ptr(0.9),
MaxTokens: 111, MaxTokens: intPtr(111),
TopP: 0.5, TopP: float64Ptr(0.5),
TimeoutSeconds: 30, TimeoutSeconds: intPtr(30),
ServiceTier: "flex", ServiceTier: "flex",
ReasoningEffort: "high", ReasoningEffort: "high",
APIKeyEnv: "RUNTIME_KEY", APIKeyEnv: "RUNTIME_KEY",
ExtraParams: map[string]string{ ExtraParams: map[string]any{
"runtime_only": "yes", "runtime_only": "yes",
}, },
} }
target := resolveExecutionTarget(profileValue, override) target, presence, err := resolveExecutionTarget(profileValue, override)
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if presence != (domain.ExecutionTargetPresence{Temperature: true, MaxTokens: true, TopP: true, TimeoutSeconds: true}) {
t.Fatalf("unexpected override presence: %+v", presence)
}
if target.Endpoint != override.Endpoint || if target.Endpoint != override.Endpoint ||
target.Model != override.Model || target.Model != override.Model ||
target.Temperature != override.Temperature || target.Temperature != *override.Temperature ||
target.MaxTokens != override.MaxTokens || target.MaxTokens != *override.MaxTokens ||
target.TopP != override.TopP || target.TopP != *override.TopP ||
target.TimeoutSeconds != override.TimeoutSeconds || target.TimeoutSeconds != *override.TimeoutSeconds ||
target.ServiceTier != override.ServiceTier || target.ServiceTier != override.ServiceTier ||
target.ReasoningEffort != override.ReasoningEffort || target.ReasoningEffort != override.ReasoningEffort ||
target.APIKeyEnv != override.APIKeyEnv { target.APIKeyEnv != override.APIKeyEnv {
@@ -1311,12 +1709,12 @@ func TestMergeExecutionTargetEmptyStringOverridesDoNotErase(t *testing.T) {
func TestMergeExecutionTargetEmptyExtraParamsDoesNotErase(t *testing.T) { func TestMergeExecutionTargetEmptyExtraParamsDoesNotErase(t *testing.T) {
base := domain.ExecutionTarget{ base := domain.ExecutionTarget{
ExtraParams: map[string]string{ ExtraParams: map[string]any{
"keep": "value", "keep": "value",
}, },
} }
override := domain.ExecutionTarget{ override := domain.ExecutionTarget{
ExtraParams: map[string]string{}, ExtraParams: map[string]any{},
} }
merged := mergeExecutionTarget(base, override) merged := mergeExecutionTarget(base, override)
@@ -1392,6 +1790,14 @@ func singleInputRef() map[string]domain.ArtifactRef {
return map[string]domain.ArtifactRef{"transcript": {Type: domain.ArtifactRefFile, URI: "a://ok"}} return map[string]domain.ArtifactRef{"transcript": {Type: domain.ArtifactRefFile, URI: "a://ok"}}
} }
func float64Ptr(v float64) *float64 {
return &v
}
func intPtr(v int) *int {
return &v
}
func newMinimalRunner(promptRepo *fakePromptRepo, execRepo *fakeExecutionProfileRepo) *Runner { func newMinimalRunner(promptRepo *fakePromptRepo, execRepo *fakeExecutionProfileRepo) *Runner {
return NewRunner( return NewRunner(
promptRepo, promptRepo,

View File

@@ -5,11 +5,14 @@ import (
"encoding/json" "encoding/json"
"errors" "errors"
"fmt" "fmt"
"io/fs"
"os" "os"
"path"
"path/filepath" "path/filepath"
"strings" "strings"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
"gitea.maximumdirect.net/eric/scriptorium/internal/filecatalog"
"github.com/santhosh-tekuri/jsonschema/v6" "github.com/santhosh-tekuri/jsonschema/v6"
) )
@@ -18,11 +21,30 @@ type StandardValidator struct {
schemaBaseDir string schemaBaseDir string
} }
type FSValidator struct {
fsys fs.FS
root string
}
func NewStandardValidator(schemaBaseDir string) Validator { func NewStandardValidator(schemaBaseDir string) Validator {
return &StandardValidator{schemaBaseDir: schemaBaseDir} return &StandardValidator{schemaBaseDir: schemaBaseDir}
} }
func NewFSValidator(fsys fs.FS, root string) Validator {
return &FSValidator{fsys: fsys, root: root}
}
func (v *StandardValidator) Validate(ctx context.Context, artifact *domain.Artifact, contract domain.OutputContract) (domain.ValidationResult, error) { func (v *StandardValidator) Validate(ctx context.Context, artifact *domain.Artifact, contract domain.OutputContract) (domain.ValidationResult, error) {
return validateArtifact(ctx, artifact, contract, v.validateJSONSchema)
}
func (v *FSValidator) Validate(ctx context.Context, artifact *domain.Artifact, contract domain.OutputContract) (domain.ValidationResult, error) {
return validateArtifact(ctx, artifact, contract, v.validateJSONSchema)
}
type schemaValidatorFunc func(instance any, schemaPath string) ([]string, error)
func validateArtifact(ctx context.Context, artifact *domain.Artifact, contract domain.OutputContract, validateSchema schemaValidatorFunc) (domain.ValidationResult, error) {
select { select {
case <-ctx.Done(): case <-ctx.Done():
return domain.ValidationResult{}, ctx.Err() return domain.ValidationResult{}, ctx.Err()
@@ -74,21 +96,14 @@ func (v *StandardValidator) Validate(ctx context.Context, artifact *domain.Artif
return res, nil return res, nil
} }
schemaPath, err := v.resolveSchemaPath(contract.SchemaPath) validationErrors, err := validateSchema(instance, contract.SchemaPath)
if err != nil { if err != nil {
return domain.ValidationResult{}, err return domain.ValidationResult{}, err
} }
if len(validationErrors) > 0 {
compiler := jsonschema.NewCompiler()
schema, err := compiler.Compile(schemaPath)
if err != nil {
return domain.ValidationResult{}, fmt.Errorf("failed to compile JSON schema %q: %w", schemaPath, err)
}
if err := schema.Validate(instance); err != nil {
res.Status = domain.ValidationFailed res.Status = domain.ValidationFailed
res.IsValid = false res.IsValid = false
res.Errors = []string{fmt.Sprintf("json schema validation failed: %v", err)} res.Errors = validationErrors
return res, nil return res, nil
} }
@@ -100,6 +115,46 @@ func (v *StandardValidator) Validate(ctx context.Context, artifact *domain.Artif
} }
} }
func (v *StandardValidator) validateJSONSchema(instance any, schemaPath string) ([]string, error) {
resolvedSchemaPath, err := v.resolveSchemaPath(schemaPath)
if err != nil {
return nil, err
}
compiler := jsonschema.NewCompiler()
schema, err := compiler.Compile(resolvedSchemaPath)
if err != nil {
return nil, fmt.Errorf("failed to compile JSON schema %q: %w", resolvedSchemaPath, err)
}
if err := schema.Validate(instance); err != nil {
return []string{fmt.Sprintf("json schema validation failed: %v", err)}, nil
}
return nil, nil
}
func (v *FSValidator) validateJSONSchema(instance any, schemaPath string) ([]string, error) {
schemaName, schemaDoc, err := v.loadSchemaDocument(schemaPath)
if err != nil {
return nil, err
}
resourceURL := fsSchemaResourceURL(schemaName)
compiler := jsonschema.NewCompiler()
if err := compiler.AddResource(resourceURL, schemaDoc); err != nil {
return nil, fmt.Errorf("failed to register JSON schema %q: %w", schemaName, err)
}
schema, err := compiler.Compile(resourceURL)
if err != nil {
return nil, fmt.Errorf("failed to compile JSON schema %q: %w", schemaName, err)
}
if err := schema.Validate(instance); err != nil {
return []string{fmt.Sprintf("json schema validation failed: %v", err)}, nil
}
return nil, nil
}
func parseJSON(body []byte) (any, error) { func parseJSON(body []byte) (any, error) {
var v any var v any
if err := json.Unmarshal(body, &v); err != nil { if err := json.Unmarshal(body, &v); err != nil {
@@ -132,6 +187,20 @@ func (v *StandardValidator) LoadSchemaDocument(ctx context.Context, schemaPath s
return doc, nil return doc, nil
} }
func (v *FSValidator) LoadSchemaDocument(ctx context.Context, schemaPath string) (any, error) {
select {
case <-ctx.Done():
return nil, ctx.Err()
default:
}
_, doc, err := v.loadSchemaDocument(schemaPath)
if err != nil {
return nil, err
}
return doc, nil
}
func (v *StandardValidator) resolveSchemaPath(schemaPath string) (string, error) { func (v *StandardValidator) resolveSchemaPath(schemaPath string) (string, error) {
if strings.TrimSpace(schemaPath) == "" { if strings.TrimSpace(schemaPath) == "" {
return "", errors.New("schema path is required for json_schema validation") return "", errors.New("schema path is required for json_schema validation")
@@ -149,3 +218,75 @@ func (v *StandardValidator) resolveSchemaPath(schemaPath string) (string, error)
return resolved, nil return resolved, nil
} }
func (v *FSValidator) loadSchemaDocument(schemaPath string) (string, any, error) {
resolved, err := v.resolveSchemaPath(schemaPath)
if err != nil {
return "", nil, err
}
raw, err := fs.ReadFile(v.fsys, resolved)
if err != nil {
return "", nil, fmt.Errorf("failed to read schema file %q: %w", resolved, err)
}
var doc any
if err := json.Unmarshal(raw, &doc); err != nil {
return "", nil, fmt.Errorf("failed to decode JSON schema %q: %w", resolved, err)
}
return resolved, doc, nil
}
func (v *FSValidator) resolveSchemaPath(schemaPath string) (string, error) {
if strings.TrimSpace(schemaPath) == "" {
return "", errors.New("schema path is required for json_schema validation")
}
if v.fsys == nil {
return "", errors.New("schema filesystem is nil")
}
cleanRoot := filecatalog.CleanFSRoot(v.root)
rootInfo, err := fs.Stat(v.fsys, cleanRoot)
if err != nil {
return "", fmt.Errorf("failed to access schema source %q: %w", cleanRoot, err)
}
var resolved string
if rootInfo.IsDir() {
resolvedPath, _, err := filecatalog.ResolveFSPath(cleanRoot, cleanRoot, schemaPath)
if err != nil {
return "", err
}
resolved = resolvedPath
} else {
cleanSchemaPath, err := cleanSchemaFSPath(schemaPath)
if err != nil {
return "", err
}
if cleanSchemaPath != path.Base(cleanRoot) {
return "", fmt.Errorf("schema path %q does not match schema file %q", cleanSchemaPath, path.Base(cleanRoot))
}
resolved = cleanRoot
}
if _, err := fs.Stat(v.fsys, resolved); err != nil {
return "", fmt.Errorf("failed to access schema file %q: %w", resolved, err)
}
return resolved, nil
}
func cleanSchemaFSPath(schemaPath string) (string, error) {
cleaned := strings.TrimSpace(schemaPath)
if cleaned == "" {
return "", errors.New("schema path is required for json_schema validation")
}
cleaned = path.Clean(cleaned)
if path.IsAbs(cleaned) {
return "", fmt.Errorf("schema path %q must be relative", schemaPath)
}
return cleaned, nil
}
func fsSchemaResourceURL(schemaName string) string {
return "scriptorium-schema:///" + strings.TrimPrefix(path.Clean(schemaName), "/")
}

View File

@@ -4,7 +4,9 @@ import (
"context" "context"
"os" "os"
"path/filepath" "path/filepath"
"strings"
"testing" "testing"
"testing/fstest"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain" "gitea.maximumdirect.net/eric/scriptorium/internal/domain"
) )
@@ -250,3 +252,132 @@ func TestStandardValidatorLoadSchemaDocumentInvalidJSON(t *testing.T) {
t.Fatal("expected decode error") t.Fatal("expected decode error")
} }
} }
func TestFSValidatorJSONSchemaSuccess(t *testing.T) {
v := NewFSValidator(fstest.MapFS{
"schemas/events.schema.json": &fstest.MapFile{Data: []byte(`{
"$schema": "https://json-schema.org/draft/2020-12/schema",
"type": "object",
"required": ["events"],
"properties": {
"events": {"type": "array"}
}
}`)},
}, "schemas")
res, err := v.Validate(context.Background(), &domain.Artifact{Body: []byte(`{"events":[]}`)}, domain.OutputContract{
ValidationMode: domain.ValidationJSONSchema,
SchemaPath: "events.schema.json",
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if res.Status != domain.ValidationPassed || !res.IsValid {
t.Fatalf("expected passed/valid, got status=%q valid=%v", res.Status, res.IsValid)
}
}
func TestFSValidatorJSONSchemaPathContainment(t *testing.T) {
t.Run("nested schema inside root succeeds", func(t *testing.T) {
v := NewFSValidator(fstest.MapFS{
"schemas/nested/events.schema.json": &fstest.MapFile{Data: []byte(`{
"type": "object",
"required": ["events"],
"properties": {
"events": {"type": "array"}
}
}`)},
}, "schemas")
res, err := v.Validate(context.Background(), &domain.Artifact{Body: []byte(`{"events":[]}`)}, domain.OutputContract{
ValidationMode: domain.ValidationJSONSchema,
SchemaPath: "nested/events.schema.json",
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if res.Status != domain.ValidationPassed || !res.IsValid {
t.Fatalf("expected passed/valid, got status=%q valid=%v", res.Status, res.IsValid)
}
})
tests := []struct {
name string
schemaPath string
wantErr string
}{
{name: "parent escape rejected", schemaPath: "../outside.schema.json", wantErr: "escapes source root"},
{name: "absolute path rejected", schemaPath: "/outside.schema.json", wantErr: "must be relative"},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
v := NewFSValidator(fstest.MapFS{
"schemas/events.schema.json": &fstest.MapFile{Data: []byte(`{"type":"object"}`)},
"outside.schema.json": &fstest.MapFile{Data: []byte(`{"type":"object"}`)},
"schemas/outside.schema.json": &fstest.MapFile{Data: []byte(`{"type":"object"}`)},
}, "schemas")
_, err := v.Validate(context.Background(), &domain.Artifact{Body: []byte(`{"events":[]}`)}, domain.OutputContract{
ValidationMode: domain.ValidationJSONSchema,
SchemaPath: tc.schemaPath,
})
if err == nil {
t.Fatal("expected schema path error")
}
if !strings.Contains(err.Error(), tc.wantErr) {
t.Fatalf("expected error to contain %q, got %v", tc.wantErr, err)
}
})
}
}
func TestFSValidatorSingleSchemaFileUsesBaseName(t *testing.T) {
v := NewFSValidator(fstest.MapFS{
"events.schema.json": &fstest.MapFile{Data: []byte(`{
"type": "object",
"required": ["events"],
"properties": {
"events": {"type": "array"}
}
}`)},
}, "events.schema.json")
res, err := v.Validate(context.Background(), &domain.Artifact{Body: []byte(`{"events":[]}`)}, domain.OutputContract{
ValidationMode: domain.ValidationJSONSchema,
SchemaPath: "events.schema.json",
})
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
if res.Status != domain.ValidationPassed || !res.IsValid {
t.Fatalf("expected passed/valid, got status=%q valid=%v", res.Status, res.IsValid)
}
_, err = v.Validate(context.Background(), &domain.Artifact{Body: []byte(`{"events":[]}`)}, domain.OutputContract{
ValidationMode: domain.ValidationJSONSchema,
SchemaPath: "other.schema.json",
})
if err == nil {
t.Fatal("expected schema path mismatch error")
}
}
func TestFSValidatorLoadSchemaDocument(t *testing.T) {
v := NewFSValidator(fstest.MapFS{
"schemas/schema.json": &fstest.MapFile{Data: []byte(`{"type":"object"}`)},
}, "schemas")
loader, ok := v.(SchemaDocumentLoader)
if !ok {
t.Fatal("fs validator must implement SchemaDocumentLoader")
}
doc, err := loader.LoadSchemaDocument(context.Background(), "schema.json")
if err != nil {
t.Fatalf("expected no error, got %v", err)
}
obj, ok := doc.(map[string]any)
if !ok || obj["type"] != "object" {
t.Fatalf("unexpected schema document: %#v", doc)
}
}

218
json_copy.go Normal file
View File

@@ -0,0 +1,218 @@
package scriptorium
import (
"encoding/json"
"fmt"
"math"
"reflect"
"strconv"
)
const maxSafeJSONInteger = 1<<53 - 1
type jsonVisit struct {
typ reflect.Type
ptr uintptr
}
func copyPublicJSONMap(src map[string]any) (map[string]any, error) {
if src == nil {
return nil, nil
}
copied, err := copyPublicJSONValue(reflect.ValueOf(src), "extra_params", make(map[jsonVisit]struct{}))
if err != nil {
return nil, err
}
out, ok := copied.(map[string]any)
if !ok {
return nil, fmt.Errorf("extra_params: expected object")
}
return out, nil
}
func copyPublicJSONValue(value reflect.Value, path string, seen map[jsonVisit]struct{}) (any, error) {
if !value.IsValid() {
return nil, nil
}
if value.Kind() == reflect.Interface {
if value.IsNil() {
return nil, nil
}
return copyPublicJSONValue(value.Elem(), path, seen)
}
if !value.CanInterface() {
return nil, fmt.Errorf("%s: value cannot be copied", path)
}
if number, ok := value.Interface().(json.Number); ok {
f, err := strconv.ParseFloat(number.String(), 64)
if err != nil || math.IsNaN(f) || math.IsInf(f, 0) {
return nil, fmt.Errorf("%s: invalid JSON number", path)
}
return number, nil
}
switch value.Kind() {
case reflect.Bool, reflect.String:
return value.Interface(), nil
case reflect.Int, reflect.Int8, reflect.Int16, reflect.Int32, reflect.Int64:
if value.Int() < -maxSafeJSONInteger || value.Int() > maxSafeJSONInteger {
return nil, fmt.Errorf("%s: integer is outside the JSON-safe range", path)
}
return value.Interface(), nil
case reflect.Uint, reflect.Uint8, reflect.Uint16, reflect.Uint32, reflect.Uint64, reflect.Uintptr:
if value.Uint() > maxSafeJSONInteger {
return nil, fmt.Errorf("%s: integer is outside the JSON-safe range", path)
}
return value.Interface(), nil
case reflect.Float32, reflect.Float64:
f := value.Convert(reflect.TypeOf(float64(0))).Float()
if math.IsNaN(f) || math.IsInf(f, 0) {
return nil, fmt.Errorf("%s: floating-point value must be finite", path)
}
return value.Interface(), nil
case reflect.Pointer:
if value.IsNil() {
return nil, nil
}
visit := jsonVisit{typ: value.Type(), ptr: value.Pointer()}
if _, ok := seen[visit]; ok {
return nil, fmt.Errorf("%s: cyclic value is not supported", path)
}
seen[visit] = struct{}{}
defer delete(seen, visit)
return copyPublicJSONValue(value.Elem(), path, seen)
case reflect.Map:
return copyPublicJSONMapValue(value, path, seen)
case reflect.Slice:
if value.IsNil() {
return nil, nil
}
return copyPublicJSONSequenceValue(value, path, seen)
case reflect.Array:
return copyPublicJSONSequenceValue(value, path, seen)
default:
return nil, fmt.Errorf("%s: unsupported JSON value type %s", path, value.Type())
}
}
func copyPublicJSONMapValue(value reflect.Value, path string, seen map[jsonVisit]struct{}) (any, error) {
if value.IsNil() {
return nil, nil
}
if value.Type().Key().Kind() != reflect.String {
return nil, fmt.Errorf("%s: map key type %s is not supported", path, value.Type().Key())
}
visit := jsonVisit{typ: value.Type(), ptr: value.Pointer()}
if _, ok := seen[visit]; ok {
return nil, fmt.Errorf("%s: cyclic value is not supported", path)
}
seen[visit] = struct{}{}
defer delete(seen, visit)
type entry struct {
key reflect.Value
name string
value any
}
entries := make([]entry, 0, value.Len())
preserveType := true
elemType := value.Type().Elem()
iter := value.MapRange()
for iter.Next() {
key := iter.Key()
name := key.String()
copied, err := copyPublicJSONValue(iter.Value(), path+"."+name, seen)
if err != nil {
return nil, err
}
entries = append(entries, entry{key: key, name: name, value: copied})
if copied == nil {
if !canAssignNil(elemType) {
preserveType = false
}
continue
}
if !reflect.TypeOf(copied).AssignableTo(elemType) {
preserveType = false
}
}
if preserveType {
out := reflect.MakeMapWithSize(value.Type(), len(entries))
for _, entry := range entries {
if entry.value == nil {
out.SetMapIndex(entry.key, reflect.Zero(elemType))
continue
}
out.SetMapIndex(entry.key, reflect.ValueOf(entry.value))
}
return out.Interface(), nil
}
out := make(map[string]any, len(entries))
for _, entry := range entries {
out[entry.name] = entry.value
}
return out, nil
}
func copyPublicJSONSequenceValue(value reflect.Value, path string, seen map[jsonVisit]struct{}) (any, error) {
var visit jsonVisit
if value.Kind() == reflect.Slice {
visit = jsonVisit{typ: value.Type(), ptr: value.Pointer()}
if _, ok := seen[visit]; ok {
return nil, fmt.Errorf("%s: cyclic value is not supported", path)
}
seen[visit] = struct{}{}
defer delete(seen, visit)
}
values := make([]any, value.Len())
preserveType := true
elemType := value.Type().Elem()
for i := 0; i < value.Len(); i++ {
copied, err := copyPublicJSONValue(value.Index(i), fmt.Sprintf("%s[%d]", path, i), seen)
if err != nil {
return nil, err
}
values[i] = copied
if copied == nil {
if !canAssignNil(elemType) {
preserveType = false
}
continue
}
if !reflect.TypeOf(copied).AssignableTo(elemType) {
preserveType = false
}
}
if preserveType {
out := reflect.New(value.Type()).Elem()
if value.Kind() == reflect.Slice {
out = reflect.MakeSlice(value.Type(), value.Len(), value.Len())
}
for i, copied := range values {
if copied == nil {
out.Index(i).Set(reflect.Zero(elemType))
continue
}
out.Index(i).Set(reflect.ValueOf(copied))
}
return out.Interface(), nil
}
out := make([]any, len(values))
copy(out, values)
return out, nil
}
func canAssignNil(typ reflect.Type) bool {
switch typ.Kind() {
case reflect.Chan, reflect.Func, reflect.Interface, reflect.Map, reflect.Pointer, reflect.Slice:
return true
default:
return false
}
}

23
llm_adapter.go Normal file
View File

@@ -0,0 +1,23 @@
package scriptorium
import (
"context"
"fmt"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
)
type publicLLMClientAdapter struct {
client LLMClient
}
func (a publicLLMClientAdapter) Generate(ctx context.Context, req domain.GenerateRequest) (*domain.GenerateResponse, error) {
resp, err := a.client.Generate(ctx, fromDomainGenerateRequest(req))
if err != nil {
return nil, err
}
if resp == nil {
return nil, fmt.Errorf("%w: llm client returned nil response", ErrLLMGenerate)
}
return toDomainGenerateResponse(resp), nil
}

124
profiles.go Normal file
View File

@@ -0,0 +1,124 @@
package scriptorium
import (
"context"
"errors"
"fmt"
"strings"
"gitea.maximumdirect.net/eric/scriptorium/internal/domain"
"gitea.maximumdirect.net/eric/scriptorium/internal/profile"
)
// OpenAICompatibleProfile returns an ordinary in-memory Profile for an
// OpenAI-compatible chat-completions endpoint.
//
// It does not register global state, maintain a model catalog, or resolve
// credentials. If APIKeyRequired is true, callers satisfy it with
// RunRequest.APIKey. Raw API keys do not belong in profiles.
func OpenAICompatibleProfile(cfg OpenAICompatibleProfileConfig) Profile {
return Profile{
ID: cfg.ID,
Endpoint: cfg.Endpoint,
Model: cfg.Model,
Temperature: cfg.Temperature,
MaxTokens: cfg.MaxTokens,
TopP: cfg.TopP,
TimeoutSeconds: cfg.TimeoutSeconds,
ServiceTier: cfg.ServiceTier,
ReasoningEffort: cfg.ReasoningEffort,
APIKeyRequired: cfg.APIKeyRequired,
ExtraParams: copyShallowAnyMap(cfg.ExtraParams),
}
}
func copyShallowAnyMap(src map[string]any) map[string]any {
if src == nil {
return nil
}
out := make(map[string]any, len(src))
for k, v := range src {
out[k] = v
}
return out
}
type memoryProfileRepository struct {
profiles map[string]domain.ExecutionProfile
}
func newMemoryProfileRepository(profiles []Profile) (*memoryProfileRepository, error) {
repo := &memoryProfileRepository{profiles: make(map[string]domain.ExecutionProfile, len(profiles))}
for _, publicProfile := range profiles {
prof, err := toDomainProfile(publicProfile)
if err != nil {
return nil, err
}
if _, exists := repo.profiles[prof.ID]; exists {
return nil, fmt.Errorf("duplicate profile id %q", prof.ID)
}
repo.profiles[prof.ID] = prof
}
return repo, nil
}
func (r *memoryProfileRepository) GetProfile(_ context.Context, id string) (*domain.ExecutionProfile, error) {
if r == nil {
return nil, profile.ErrProfileNotFound
}
prof, ok := r.profiles[id]
if !ok {
return nil, profile.ErrProfileNotFound
}
prof.ExtraParams = copyAnyMap(prof.ExtraParams)
return &prof, nil
}
func toDomainProfile(publicProfile Profile) (domain.ExecutionProfile, error) {
extraParams, err := copyPublicJSONMap(publicProfile.ExtraParams)
if err != nil {
return domain.ExecutionProfile{}, err
}
prof := domain.ExecutionProfile{
ID: strings.TrimSpace(publicProfile.ID),
Endpoint: publicProfile.Endpoint,
Model: publicProfile.Model,
Temperature: publicProfile.Temperature,
MaxTokens: publicProfile.MaxTokens,
TopP: publicProfile.TopP,
TimeoutSeconds: publicProfile.TimeoutSeconds,
ServiceTier: publicProfile.ServiceTier,
ReasoningEffort: publicProfile.ReasoningEffort,
APIKeyRequired: publicProfile.APIKeyRequired,
ExtraParams: extraParams,
}
if err := validatePublicProfile(prof); err != nil {
return domain.ExecutionProfile{}, err
}
return prof, nil
}
func validatePublicProfile(prof domain.ExecutionProfile) error {
if strings.TrimSpace(prof.ID) == "" {
return errors.New("id is required")
}
if strings.TrimSpace(prof.Endpoint) == "" {
return errors.New("endpoint is required")
}
if strings.TrimSpace(prof.Model) == "" {
return errors.New("model is required")
}
if prof.Temperature < 0 || prof.Temperature > 2 {
return errors.New("temperature must be between 0 and 2")
}
if prof.MaxTokens < 0 {
return errors.New("max_tokens must be greater than or equal to 0")
}
if prof.TopP < 0 || prof.TopP > 1 {
return errors.New("top_p must be between 0 and 1")
}
if prof.TimeoutSeconds < 0 {
return errors.New("timeout_seconds must be greater than or equal to 0")
}
return nil
}

298
types.go Normal file
View File

@@ -0,0 +1,298 @@
package scriptorium
import (
"context"
"time"
)
// ArtifactRefType defines how an artifact is referenced.
type ArtifactRefType string
const (
ArtifactRefInline ArtifactRefType = "inline"
ArtifactRefFile ArtifactRefType = "file"
)
// OutputFormat defines the desired output format.
type OutputFormat string
const (
FormatText OutputFormat = "text"
FormatMarkdown OutputFormat = "markdown"
FormatJSON OutputFormat = "json"
)
// ValidationMode defines the output validation strategy.
type ValidationMode string
const (
ValidationNone ValidationMode = "none"
ValidationBasic ValidationMode = "basic"
ValidationJSON ValidationMode = "json"
ValidationJSONSchema ValidationMode = "json_schema"
)
// ValidationStatus defines the result of a validation check.
type ValidationStatus string
const (
ValidationPassed ValidationStatus = "passed"
ValidationFailed ValidationStatus = "failed"
ValidationSkipped ValidationStatus = "skipped"
)
// CacheControlType defines provider cache behavior for prompt content.
type CacheControlType string
const (
CacheControlEphemeral CacheControlType = "ephemeral"
)
// StructuredOutputType identifies provider-level structured output modes.
type StructuredOutputType string
const (
StructuredOutputJSONSchema StructuredOutputType = "json_schema"
)
// RunRequest represents a request to prepare or run a single prompt.
type RunRequest struct {
PromptID string
PromptVersion string
ProfileID string
APIKey string `json:"-"`
Inputs map[string]ArtifactRef
Vars map[string]string
Execution *ExecutionTargetOverride
Validation *OutputContract
Metadata map[string]string
}
// PreparedRun contains prepared prompt execution state. It does not include
// resolved API key values, model output, validation results, or internal target
// presence metadata.
type PreparedRun struct {
PromptID string `json:"prompt_id"`
PromptVersion string `json:"prompt_version,omitempty"`
PromptHash string `json:"prompt_hash,omitempty"`
SelectedProfileID string `json:"selected_profile_id"`
EffectiveModelParams ExecutionTarget `json:"effective_model_params"`
OutputContract OutputContract `json:"output_contract"`
StructuredOutput *StructuredOutputSpec `json:"structured_output,omitempty"`
InputHashes map[string]string `json:"input_hashes,omitempty"`
SessionID string `json:"session_id,omitempty"`
RenderedPromptHash string `json:"rendered_prompt_hash"`
Messages []RenderedMessage `json:"messages"`
StartTime time.Time `json:"start_time,omitempty"`
EndTime time.Time `json:"end_time,omitempty"`
DurationMS int64 `json:"duration_ms,omitempty"`
}
// RunResult contains generated output, validation state, and run metadata.
type RunResult struct {
RunID string `json:"run_id"`
Artifact Artifact `json:"artifact"`
RawOutput string `json:"raw_output"`
Validation ValidationResult `json:"validation"`
PromptID string `json:"prompt_id"`
PromptVersion string `json:"prompt_version,omitempty"`
PromptHash string `json:"prompt_hash,omitempty"`
RenderedPromptHash string `json:"rendered_prompt_hash"`
SelectedProfileID string `json:"selected_profile_id"`
ModelName string `json:"model_name"`
Endpoint string `json:"endpoint"`
EffectiveModelParams ExecutionTarget `json:"effective_model_params"`
InputHashes map[string]string `json:"input_hashes,omitempty"`
Usage TokenUsage `json:"usage"`
StartTime time.Time `json:"start_time,omitempty"`
EndTime time.Time `json:"end_time,omitempty"`
Duration time.Duration `json:"duration,omitempty"`
}
// ArtifactRef represents a reference to prompt input content.
type ArtifactRef struct {
Type ArtifactRefType
URI string
Body string
}
// Artifact represents loaded artifact content.
type Artifact struct {
Name string
ContentType string
Body []byte
URI string
Size int64
Hash string
}
// ExecutionTarget represents effective model runtime settings.
type ExecutionTarget struct {
Endpoint string `json:"endpoint"`
Model string `json:"model"`
Temperature float64 `json:"temperature"`
MaxTokens int `json:"max_tokens"`
TopP float64 `json:"top_p"`
TimeoutSeconds int `json:"timeout_seconds"`
ServiceTier string `json:"service_tier"`
ReasoningEffort string `json:"reasoning_effort"`
APIKeyEnv string `json:"api_key_env"`
ExtraParams map[string]any `json:"extra_params"`
}
// ExecutionTargetOverride represents per-request runtime setting overrides.
type ExecutionTargetOverride struct {
Endpoint string
Model string
Temperature *float64
MaxTokens *int
TopP *float64
TimeoutSeconds *int
ServiceTier string
ReasoningEffort string
APIKeyEnv string
ExtraParams map[string]any
}
// Profile is an in-memory execution profile for library consumers.
//
// It is equivalent to a loaded profile file after validation. Raw API keys do
// not belong in profiles; use APIKeyRequired to require callers to provide
// RunRequest.APIKey for each request, or use profile YAML api_key_env with file
// and FS profile sources.
type Profile struct {
ID string
Endpoint string
Model string
Temperature float64
MaxTokens int
TopP float64
TimeoutSeconds int
ServiceTier string
ReasoningEffort string
APIKeyRequired bool
ExtraParams map[string]any
}
// OpenAICompatibleProfileConfig configures an OpenAI-compatible in-memory
// profile.
//
// It contains ordinary profile fields for OpenAI-compatible chat-completions
// endpoints. APIKeyRequired is satisfied by RunRequest.APIKey. Raw API keys do
// not belong in this config.
type OpenAICompatibleProfileConfig struct {
ID string
Endpoint string
Model string
APIKeyRequired bool
Temperature float64
MaxTokens int
TopP float64
TimeoutSeconds int
ServiceTier string
ReasoningEffort string
ExtraParams map[string]any
}
// ExecutionTargetPresence tracks which numeric runtime settings were explicit
// request overrides.
type ExecutionTargetPresence struct {
Temperature bool
MaxTokens bool
TopP bool
TimeoutSeconds bool
}
// OutputContract defines output and validation requirements.
type OutputContract struct {
Format OutputFormat `json:"format"`
ValidationMode ValidationMode `json:"validation_mode"`
SchemaPath string `json:"schema_path"`
RepairAttempts int `json:"repair_attempts"`
}
// ValidationResult represents output validation state.
type ValidationResult struct {
Status ValidationStatus `json:"status"`
Mode ValidationMode `json:"mode"`
Errors []string `json:"errors,omitempty"`
SchemaPath string `json:"schema_path,omitempty"`
RepairAttempts int `json:"repair_attempts"`
IsValid bool `json:"is_valid"`
}
// TokenUsage tracks token consumption.
type TokenUsage struct {
PromptTokens int `json:"prompt_tokens"`
CompletionTokens int `json:"completion_tokens"`
TotalTokens int `json:"total_tokens"`
CachedTokens int `json:"cached_tokens"`
CacheWriteTokens int `json:"cache_write_tokens"`
}
// RenderedPrompt is the fully rendered prompt passed to an LLM client.
type RenderedPrompt struct {
SessionID string `json:"session_id,omitempty"`
Messages []RenderedMessage `json:"messages"`
}
// RenderedMessage is a rendered chat message.
type RenderedMessage struct {
Role string `json:"role"`
Content string `json:"content"`
CacheControl *CacheControl `json:"cache_control,omitempty"`
}
// CacheControl describes provider cache metadata attached to prompt content.
type CacheControl struct {
Type CacheControlType `json:"type"`
TTL string `json:"ttl,omitempty"`
}
// StructuredOutputSpec describes provider-level structured output.
type StructuredOutputSpec struct {
Type StructuredOutputType `json:"type"`
JSONSchema *StructuredOutputJSONSpec `json:"json_schema,omitempty"`
}
// StructuredOutputJSONSpec contains JSON Schema output constraints.
type StructuredOutputJSONSpec struct {
Name string `json:"name"`
Strict bool `json:"strict"`
Schema any `json:"schema"`
}
// LLMClient executes rendered prompts for Engine.Run.
type LLMClient interface {
Generate(context.Context, GenerateRequest) (*GenerateResponse, error)
}
// GenerateRequest is passed to an injected LLM client.
type GenerateRequest struct {
Prompt RenderedPrompt `json:"prompt"`
Target ExecutionTarget `json:"target"`
TargetPresence ExecutionTargetPresence `json:"target_presence"`
StructuredOutput *StructuredOutputSpec `json:"structured_output,omitempty"`
APIKey string `json:"-"`
}
// GenerateResponse is returned by an injected LLM client.
type GenerateResponse struct {
Content string `json:"content"`
Usage TokenUsage `json:"usage"`
}
// File returns a file-backed artifact reference.
func File(path string) ArtifactRef {
return ArtifactRef{Type: ArtifactRefFile, URI: path}
}
// Inline returns an inline artifact reference.
func Inline(body string) ArtifactRef {
return ArtifactRef{Type: ArtifactRefInline, Body: body}
}
// InlineWithURI returns an inline artifact reference with URI metadata.
func InlineWithURI(uri string, body string) ArtifactRef {
return ArtifactRef{Type: ArtifactRefInline, URI: uri, Body: body}
}