119 Commits

Author SHA1 Message Date
a4264bf4b4 Prepare the v0.9.0 release 2026-08-26 17:31:53 +00:00
81f41564a2 Harden external catalog validation 2026-08-26 16:51:31 +00:00
6b5a2497cc Document the external catalog boundary 2026-08-26 15:06:33 +00:00
422cc6c978 Use external catalogs for maintained backends 2026-08-26 15:04:50 +00:00
f48e042565 Validate the external backend catalogs 2026-08-26 15:00:50 +00:00
d9442850ef Add immutable profile catalog loading 2026-08-26 14:56:56 +00:00
2e828157b6 Mark the Rakestrawhome catalog release complete 2026-08-26 14:54:42 +00:00
4189146536 Mark the OpenRouter catalog release complete 2026-08-26 14:53:02 +00:00
59be507d2f Freeze the built-in catalog compatibility baseline 2026-08-26 14:49:25 +00:00
b45cea4084 Refine the external catalog implementation plan 2026-08-26 14:20:22 +00:00
efe885c6b2 Plan external catalog extraction 2026-08-26 13:58:22 +00:00
404a4c331d Clarify appended messages and streamline repair 2026-08-26 02:37:47 +00:00
174516cb39 Document appended request messages 2026-08-26 02:19:44 +00:00
b0999112cc Compose appended request messages 2026-08-26 02:16:16 +00:00
ed0f7527d5 Add appended message request validation 2026-08-26 02:10:01 +00:00
5064cf833d Centralize message role invariants 2026-08-26 02:07:15 +00:00
8745d256bd Plan appended request messages 2026-08-26 01:53:22 +00:00
2e76003fd5 Prepare documentation for the v0.8.0 release 2026-08-25 12:19:17 +00:00
36ce5a5099 Fix output repair diagnostics and concurrency tests 2026-08-25 11:54:50 +00:00
465dc1389d Document bounded output repair 2026-08-25 09:58:10 +00:00
ae6f1a9865 Enable bounded output repair in the engine 2026-08-25 09:55:19 +00:00
ee99dc9478 Run bounded output repair attempts 2026-08-25 09:51:28 +00:00
00ee5893e9 Build bounded output repair requests 2026-08-25 09:46:56 +00:00
64d1cffd89 Preserve explicit empty model responses 2026-08-25 09:42:28 +00:00
f9e8afa2c3 Enforce bounded output repair contracts 2026-08-25 09:41:15 +00:00
e827631d8c Plan the public output repair implementation 2026-08-25 09:38:06 +00:00
115fe8ba58 Plan bounded output repair 2026-08-25 03:33:33 +00:00
c53250f023 Prepare documentation for the v0.7.0 release 2026-08-25 02:13:43 +00:00
3d99483219 Preserve cancellation after profile lookups 2026-08-25 02:06:47 +00:00
67f788b1e2 Document profile inheritance behavior 2026-08-25 01:46:16 +00:00
764103a2e2 Resolve inherited profiles in engine workflows 2026-08-25 01:41:21 +00:00
a08dd83d1f Add profile inheritance resolver 2026-08-25 01:38:15 +00:00
e8922d8ec5 Add profile inheritance definition support 2026-08-25 01:34:40 +00:00
2d44305a8a Document optional API key environment behavior 2026-08-25 00:44:26 +00:00
a11c80291e Allow unauthenticated optional API key requests 2026-08-25 00:40:49 +00:00
3239567297 Make optional API key environments nonblocking 2026-08-25 00:39:26 +00:00
c239304c2a Document structured generation errors 2026-08-23 19:04:00 +00:00
159b02116f Expose structured generation errors to consumers 2026-08-23 19:01:38 +00:00
e5b7adfb49 Return structured errors for provider status failures 2026-08-23 18:58:15 +00:00
fa2384e696 Bound provider error response body reads 2026-08-23 18:56:15 +00:00
af0bd3f31a Add internal structured provider error parsing 2026-08-23 18:54:42 +00:00
5fff8cd623 Retire the completed Rakestrawhome backend roadmaps 2026-08-23 17:34:14 +00:00
b2a6c47778 Document the Rakestrawhome built-in backend 2026-08-23 17:19:12 +00:00
a1805fe550 Add the Rakestrawhome Gemma profile 2026-08-23 17:12:17 +00:00
93af155254 Add the Rakestrawhome built-in backend 2026-08-23 17:10:34 +00:00
d783b687a5 Plan the built-in Rakestrawhome backend 2026-08-23 17:00:01 +00:00
4ca3be2c14 Finish audit remediation and prepare v0.6.0 2026-08-12 13:01:05 +00:00
227fb35f99 Centralize the maintainer validation workflow 2026-08-12 00:04:48 +00:00
e291b8bfe9 Consolidate provider transport test scaffolding 2026-08-11 23:58:31 +00:00
2b6a7f83c4 Bound and strictly decode provider responses 2026-08-11 23:46:16 +00:00
3a43550f70 Validate and compose provider endpoints 2026-08-11 23:38:45 +00:00
c281f721bc Preserve transport error identities 2026-08-11 23:28:00 +00:00
350b0e76d9 Preserve repair settings and cumulative usage 2026-08-11 23:21:34 +00:00
e43350fd0d Make validation cancellation authoritative 2026-08-11 23:14:01 +00:00
20d3e3b5ee Escape schema resources and reuse compiled plans 2026-08-11 23:05:11 +00:00
a93b799236 Preserve exact JSON validation semantics 2026-08-11 22:55:17 +00:00
e83a3ce179 Honor cancellation while rendering prompts 2026-08-11 22:48:20 +00:00
a04a3bbc5f Correct artifact file and empty input handling 2026-08-11 22:41:37 +00:00
731b66cff5 Avoid decoding unrelated profiles 2026-08-11 22:32:33 +00:00
70e0ea0cf0 Correct profile source validation and identity 2026-08-11 22:28:48 +00:00
d45c474c1e Unify prompt repository source handling 2026-08-11 22:18:50 +00:00
25f1ba0b30 Correct prompt definition selection and decoding 2026-08-11 22:11:39 +00:00
a718762da1 Contain prompt content paths within source roots 2026-08-11 22:03:32 +00:00
58ac3ce298 Correct engine construction edge cases 2026-08-11 21:54:47 +00:00
57f2ce1ce4 Harden public ownership and diagnostic contracts 2026-08-11 21:45:54 +00:00
c8b6d5c490 Consolidate public JSON serialization 2026-08-11 21:40:12 +00:00
abeb50b525 Bound JSON-compatible value copying 2026-08-11 21:32:24 +00:00
1cb07c7d91 Centralize output contract validation 2026-08-11 21:21:33 +00:00
8cfc71c351 Centralize execution setting and session validation 2026-08-11 21:12:56 +00:00
5ccfa4a345 Plan the codebase audit remediation 2026-08-11 20:10:14 +00:00
14e03f19d0 Consolidate and close the codebase audit 2026-08-11 17:21:29 +00:00
7c562a9374 Document cross-cutting architecture audit findings 2026-08-11 17:10:59 +00:00
ef97d85ac9 Document repository-wide test strategy audit 2026-08-11 17:00:59 +00:00
805e48c873 Document capacity scheduling audit results 2026-08-11 16:51:28 +00:00
e9e126dcba Document OpenAI transport audit findings 2026-08-11 16:42:55 +00:00
32e7a3557c Document prepared execution lifecycle audit 2026-08-11 16:31:05 +00:00
1d1b04e2e0 Record ordinary execution and repair audit findings 2026-08-11 16:22:09 +00:00
9748897751 Document inspection and target resolution audit findings 2026-08-11 16:09:50 +00:00
9d020039d5 Record output validation audit findings 2026-08-11 15:59:46 +00:00
5247ce0b73 Record artifact loading and prompt rendering audit findings 2026-08-11 15:41:57 +00:00
c434aa1dae Record profile source audit findings 2026-08-11 15:29:23 +00:00
ac9b3f3d80 Record prompt source audit findings 2026-08-11 15:18:17 +00:00
4f12a89a1b Record backend registry and defaults audit findings 2026-08-11 15:05:15 +00:00
df31e7f58e Record domain and JSON value audit findings 2026-08-11 14:54:56 +00:00
0678d242b9 Record engine operation audit findings 2026-08-11 14:43:20 +00:00
3b4ea21208 Record engine construction audit findings 2026-08-11 14:36:33 +00:00
1430e85147 Record configuration and adapter audit findings 2026-08-11 14:28:07 +00:00
34d7a19da5 Record public value and error audit findings 2026-08-11 14:17:55 +00:00
ebf1602635 Prepare the codebase audit plan 2026-08-11 14:05:30 +00:00
31f2ce3a09 Document Promptkit v0.5.0 2026-08-01 13:18:46 +00:00
fd06e4ca6b Clean up roadmap and fallback profile guidance 2026-08-01 13:16:26 +00:00
e63b8de1e9 Complete application fallback profile implementation 2026-08-01 12:39:01 +00:00
9354d2b373 Add application fallback profile sources 2026-08-01 12:37:01 +00:00
01ca5430bd Move profile composition to the engine facade 2026-08-01 12:31:50 +00:00
ae2179d103 Complete optional parameter omission 2026-08-01 02:43:02 +00:00
a248433d0f Omit unset optional request parameters 2026-08-01 02:41:44 +00:00
bd6cffc9d0 Prepare documentation for Promptkit v0.4.0 2026-07-30 23:48:25 +00:00
e40c4f182b Document structured capacity errors 2026-07-30 23:26:44 +00:00
7428e50c2c Expose structured capacity errors 2026-07-30 23:24:04 +00:00
63c67a4520 Add internal capacity error identity 2026-07-30 23:21:27 +00:00
25a7052a3d Clarify prompt input requirement documentation 2026-07-30 22:10:33 +00:00
fc3255967e Document prompt inspection API 2026-07-30 21:06:33 +00:00
e920168b30 Expose prompt inspection through the engine 2026-07-30 21:03:47 +00:00
272b6a4bc1 Add internal prompt inspection 2026-07-30 21:00:05 +00:00
dde48a31fc Document profile inspection API 2026-07-30 19:57:07 +00:00
242eace4a7 Expose profile inspection through the engine 2026-07-30 19:52:00 +00:00
0bf5f88136 Add internal profile inspection resolution 2026-07-30 19:48:09 +00:00
369ab5392d Add feature roadmap and implementation plan for profile inspection API 2026-07-30 19:42:51 +00:00
2ba0146e5d Tighten prepared execution credential handling 2026-07-30 19:06:17 +00:00
6112c2af0c Document prepared execution workflow 2026-07-30 18:27:03 +00:00
f5e12c00f5 Expose prepared execution handles 2026-07-30 18:20:35 +00:00
49fe402dd2 Add prepared execution lifecycle to the runner 2026-07-30 18:10:28 +00:00
c301eb8d55 Add frozen validation preparation plans 2026-07-30 17:59:45 +00:00
c13e9710d9 Add feature roadmap and implementation plan for downstream consumer wishlist items 2026-07-30 17:54:44 +00:00
87b5ec3d75 Organize downstream feature requests in the future roadmap 2026-07-30 17:13:26 +00:00
cb4028a637 Add feature roadmaps with wishlists from downstream consumers 2026-07-30 16:59:37 +00:00
5a1bff4529 Prepare the local backend convenience release 2026-07-30 04:01:19 +00:00
805a7f965d Document local backend configuration paths 2026-07-30 03:40:52 +00:00
147f5e5ff5 Add local backend convenience constructor 2026-07-30 03:38:23 +00:00
142 changed files with 18731 additions and 4550 deletions

View File

@@ -33,6 +33,27 @@ boundary and constraints that framework work must preserve.
## Release Guidance ## Release Guidance
Consumers upgrading from `v0.8.0` to `v0.9.0` should read the
[v0.9.0 changelog and migration guide](docs/releases/v0.9.0.md).
Consumers upgrading from `v0.7.0` to `v0.8.0` can consult the
[v0.8.0 changelog and migration guide](docs/releases/v0.8.0.md).
Earlier adopters can consult the
[v0.7.0 changelog and migration guide](docs/releases/v0.7.0.md).
Consumers upgrading from `v0.5.0` to `v0.6.0` can consult the
[v0.6.0 changelog and migration guide](docs/releases/v0.6.0.md).
Consumers upgrading from `v0.4.0` to `v0.5.0` can consult the
[v0.5.0 changelog and migration guide](docs/releases/v0.5.0.md).
Consumers upgrading from `v0.3.0` to `v0.4.0` should read the
[v0.4.0 changelog and adoption guide](docs/releases/v0.4.0.md).
Consumers upgrading from `v0.2.0` to `v0.3.0` should read the
[v0.3.0 changelog](docs/releases/v0.3.0.md).
Consumers moving from `v0.1.0` to `v0.2.0` should read the Consumers moving from `v0.1.0` to `v0.2.0` should read the
[v0.2.0 changelog and migration guide](docs/releases/v0.2.0.md). [v0.2.0 changelog and migration guide](docs/releases/v0.2.0.md).

View File

@@ -0,0 +1,145 @@
package promptkit_test
import (
"context"
"errors"
"reflect"
"sync/atomic"
"testing"
"gitea.maximumdirect.net/eric/promptkit"
)
func TestAppendedMessageRoleConstantsAreStrings(t *testing.T) {
var (
developer string = promptkit.RoleDeveloper
system string = promptkit.RoleSystem
user string = promptkit.RoleUser
assistant string = promptkit.RoleAssistant
)
if developer != "developer" || system != "system" || user != "user" || assistant != "assistant" {
t.Fatalf("unexpected role constants: %q %q %q %q", developer, system, user, assistant)
}
}
func TestAppendedMessagesComposeFrozenEffectivePrompt(t *testing.T) {
cacheControl := &promptkit.CacheControl{Type: promptkit.CacheControlEphemeral, TTL: "1h"}
appended := []promptkit.RenderedMessage{
{Role: " \tAsSiStAnT\n", Content: "previous response"},
{Role: promptkit.RoleUser, Content: "corrective request", CacheControl: cacheControl},
}
client := &fakeLLMClient{response: &promptkit.GenerateResponse{Content: "output"}}
engine, err := promptkit.NewEngine(
promptkit.Config{},
promptkit.WithPromptFS(contractPromptFS("prompt", "profile", "configured prompt"), "."),
promptkit.WithProfiles(promptkit.Profile{ID: "profile", Endpoint: "http://example.test/v1", Model: "model"}),
promptkit.WithLLMClient(client),
)
if err != nil {
t.Fatalf("construct engine: %v", err)
}
request := promptkit.RunRequest{PromptID: "prompt", AppendedMessages: appended}
plain, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: "prompt"})
if err != nil {
t.Fatalf("prepare plain request: %v", err)
}
prepared, err := engine.Prepare(context.Background(), request)
if err != nil {
t.Fatalf("prepare appended request: %v", err)
}
expected := []promptkit.RenderedMessage{
{Role: promptkit.RoleUser, Content: "configured prompt"},
{Role: promptkit.RoleAssistant, Content: "previous response"},
{Role: promptkit.RoleUser, Content: "corrective request", CacheControl: &promptkit.CacheControl{Type: promptkit.CacheControlEphemeral, TTL: "1h"}},
}
if !reflect.DeepEqual(prepared.Messages, expected) || prepared.PromptHash != plain.PromptHash || prepared.RenderedPromptHash == plain.RenderedPromptHash {
t.Fatalf("prepared effective prompt = %#v, plain = %#v", prepared, plain)
}
equivalent, err := engine.Prepare(context.Background(), promptkit.RunRequest{PromptID: "prompt", AppendedMessages: []promptkit.RenderedMessage{
{Role: promptkit.RoleAssistant, Content: "previous response"},
{Role: promptkit.RoleUser, Content: "corrective request", CacheControl: &promptkit.CacheControl{Type: promptkit.CacheControlEphemeral, TTL: "1h"}},
}})
if err != nil || equivalent.RenderedPromptHash != prepared.RenderedPromptHash {
t.Fatalf("normalized-equivalent request = (%#v, %v), want matching rendered hash %q", equivalent, err, prepared.RenderedPromptHash)
}
for _, variant := range []promptkit.RunRequest{
{PromptID: "prompt", AppendedMessages: []promptkit.RenderedMessage{{Role: promptkit.RoleAssistant, Content: "changed"}, expected[2]}},
{PromptID: "prompt", AppendedMessages: []promptkit.RenderedMessage{expected[2], expected[1]}},
{PromptID: "prompt", AppendedMessages: []promptkit.RenderedMessage{{Role: promptkit.RoleAssistant, Content: "previous response"}, {Role: promptkit.RoleUser, Content: "corrective request"}}},
} {
variantPrepared, prepareErr := engine.Prepare(context.Background(), variant)
if prepareErr != nil || variantPrepared.RenderedPromptHash == prepared.RenderedPromptHash || variantPrepared.PromptHash != prepared.PromptHash {
t.Fatalf("variant preparation = (%#v, %v)", variantPrepared, prepareErr)
}
}
result, err := engine.Run(context.Background(), request)
if err != nil || result.RenderedPromptHash != prepared.RenderedPromptHash || !reflect.DeepEqual(client.requests[0].Prompt.Messages, expected) {
t.Fatalf("run result = (%#v, %v), request = %#v", result, err, client.requests)
}
execution, err := engine.PrepareExecution(context.Background(), request)
if err != nil {
t.Fatalf("prepare execution: %v", err)
}
appended[0].Content = "changed caller content"
cacheControl.TTL = ""
firstDetails := execution.Details()
firstDetails.Messages[1].Content = "changed details content"
firstDetails.Messages[2].CacheControl.Type = "changed details cache"
secondDetails := execution.Details()
if !reflect.DeepEqual(secondDetails.Messages, expected) || secondDetails.RenderedPromptHash != prepared.RenderedPromptHash {
t.Fatalf("prepared execution details = %#v, want frozen %#v", secondDetails, expected)
}
result, err = engine.RunPrepared(context.Background(), execution)
if err != nil || result.RenderedPromptHash != prepared.RenderedPromptHash || !reflect.DeepEqual(client.requests[1].Prompt.Messages, expected) {
t.Fatalf("prepared result = (%#v, %v), requests = %#v", result, err, client.requests)
}
}
func TestInvalidAppendedMessagesFailBeforeSourceOrModelWork(t *testing.T) {
promptSource := &inspectionCountingFS{}
var modelCalls atomic.Int64
engine, err := promptkit.NewEngine(
promptkit.Config{},
promptkit.WithPromptFS(promptSource, "."),
promptkit.WithLLMClient(countingLLMClient{calls: &modelCalls}),
)
if err != nil {
t.Fatalf("construct engine: %v", err)
}
request := promptkit.RunRequest{
PromptID: "unreached",
AppendedMessages: []promptkit.RenderedMessage{{
Role: "unsupported-role",
Content: "sensitive appended content",
}},
}
operations := []struct {
name string
run func() error
}{
{name: "Prepare", run: func() error { _, err := engine.Prepare(context.Background(), request); return err }},
{name: "PrepareExecution", run: func() error { _, err := engine.PrepareExecution(context.Background(), request); return err }},
{name: "Run", run: func() error { _, err := engine.Run(context.Background(), request); return err }},
}
for _, operation := range operations {
t.Run(operation.name, func(t *testing.T) {
err := operation.run()
if !errors.Is(err, promptkit.ErrInvalidRequest) {
t.Fatalf("error = %v, want ErrInvalidRequest", err)
}
})
}
if promptSource.opens.Load() != 0 {
t.Fatalf("invalid appended message opened prompt sources %d times", promptSource.opens.Load())
}
if modelCalls.Load() != 0 {
t.Fatalf("invalid appended message invoked the model %d times", modelCalls.Load())
}
}

View File

@@ -9,50 +9,83 @@ import (
// backend. // backend.
const BackendOpenRouter = backend.OpenRouterID const BackendOpenRouter = backend.OpenRouterID
// BackendRakestrawHome is the reserved ID of Promptkit's built-in
// Rakestrawhome backend.
const BackendRakestrawHome = backend.RakestrawHomeID
// BackendLocal is the case-sensitive conventional ID used by [LocalBackend].
// It is not a built-in or reserved backend and must be registered with
// [WithBackend].
const BackendLocal = "local"
// Backend configures one engine-scoped OpenAI-compatible backend. // Backend configures one engine-scoped OpenAI-compatible backend.
// //
// Backend has no stable JSON representation. Use keyed literals so additions // Backend has no stable JSON representation. Use keyed literals so additions
// to this configuration value do not break source compatibility. // to this configuration value do not break source compatibility.
type Backend struct { type Backend struct {
// ID is the stable, case-sensitive registry key. NewEngine trims it and // ID is the stable, case-sensitive registry key. NewEngine trims it and
// requires a non-blank value. BackendOpenRouter is reserved. // requires a non-blank value. Built-in backend IDs are reserved.
ID string ID string
// Endpoint is the OpenAI-compatible base endpoint. NewEngine trims it and // Endpoint is the OpenAI-compatible base endpoint. NewEngine trims it and
// requires an absolute HTTP or HTTPS URL with a host and without user // requires an absolute HTTP or HTTPS URL with a host and without user
// information, a query string, or a fragment. Paths are allowed. // information, a query string, or a fragment. Paths are allowed.
Endpoint string Endpoint string
// APIKeyEnv optionally names the environment variable containing the API // APIKeyEnv optionally names an environment lookup source for an API key.
// key. NewEngine trims it and requires the portable form // NewEngine trims it and requires the portable form [A-Za-z_][A-Za-z0-9_]*.
// [A-Za-z_][A-Za-z0-9_]*. Store only the name, never a credential value. // A direct RunRequest.APIKey takes precedence. When no usable credential is
// available, the built-in client omits Authorization; injected clients own
// their own credential-resolution behavior. Store only the name, never a
// credential value.
APIKeyEnv string APIKeyEnv string
// ExtraParams contains backend-wide request defaults. Values must be // ExtraParams contains backend-wide request defaults. Values must be
// JSON-compatible, finite, acyclic, and keyed by non-empty strings. Keys // JSON-compatible, finite, acyclic, and keyed by non-empty strings. Keys
// must not be model, session_id, messages, temperature, max_tokens, top_p, // must not be model, session_id, messages, temperature, max_tokens, top_p,
// service_tier, reasoning_effort, or response_format. An empty map supplies // service_tier, reasoning_effort, or response_format. An empty map supplies
// no defaults. NewEngine deeply copies the map. // no defaults. NewEngine deeply copies the map and rejects excessively deep
// or large values for safety.
ExtraParams map[string]any ExtraParams map[string]any
// ConcurrencyLimit is the maximum number of simultaneous model-generation // ConcurrencyLimit is the maximum number of simultaneous model-generation
// calls allowed for this backend within one Engine. Zero leaves the backend // calls allowed for this backend within one Engine. Zero leaves the backend
// unlimited. A negative value makes NewEngine fail with ErrInvalidConfig. // unlimited. A negative value makes NewEngine fail with ErrInvalidConfig.
ConcurrencyLimit int ConcurrencyLimit int
// QueueCapacity controls how many additional Run calls may be admitted // QueueCapacity controls how many additional Run or RunPrepared calls may
// beyond ConcurrencyLimit. Nil uses 1024 when ConcurrencyLimit is positive; // be admitted beyond ConcurrencyLimit. Nil uses 1024 when ConcurrencyLimit
// a pointer uses its exact value, including zero. The pointed-to value must // is positive; a pointer uses its exact value, including zero. The pointed-to
// be non-negative, and QueueCapacity must be nil when ConcurrencyLimit is // value must be non-negative, and QueueCapacity must be nil when
// zero. Their sum must fit in an int. WithBackend copies the value and does // ConcurrencyLimit is zero. Their sum must fit in an int. WithBackend copies
// not retain the pointer. // the value and does not retain the pointer.
QueueCapacity *int QueueCapacity *int
} }
// LocalBackend returns a caller-owned Backend for a conventional local
// OpenAI-compatible endpoint. It sets ID to BackendLocal and copies endpoint
// and concurrencyLimit into Endpoint and ConcurrencyLimit without
// normalization or validation. APIKeyEnv, ExtraParams, and QueueCapacity keep
// their zero values.
//
// LocalBackend does not read environment variables, register the value, or
// mutate engine or package state. Supply the returned value through
// [WithBackend]; [NewEngine] then applies the ordinary backend validation and
// concurrency semantics, including default queue capacity for a positive
// limit, unlimited behavior for zero, and ErrInvalidConfig for a negative
// limit.
func LocalBackend(endpoint string, concurrencyLimit int) Backend {
return Backend{
ID: BackendLocal,
Endpoint: endpoint,
ConcurrencyLimit: concurrencyLimit,
}
}
// WithBackend adds one Backend registration to the constructed Engine. // WithBackend adds one Backend registration to the constructed Engine.
// //
// Registrations accumulate in option order. Every normalized ID must be unique // Registrations accumulate in option order. Every normalized ID must be unique
// across consumer registrations and built-ins; a duplicate or invalid // across consumer registrations and built-ins; a duplicate or invalid
// definition makes NewEngine fail with ErrInvalidConfig. In particular, // definition makes NewEngine fail with ErrInvalidConfig. Built-in IDs,
// BackendOpenRouter cannot be replaced. The immutable registration is scoped // including [BackendOpenRouter] and [BackendRakestrawHome], cannot be
// to the resulting Engine and cannot be enumerated, replaced, removed, or // replaced. The immutable registration is scoped to the resulting Engine and
// mutated after construction. WithBackend does not install package-global // cannot be enumerated, replaced, removed, or mutated after construction.
// state. // WithBackend does not install package-global state.
func WithBackend(backend Backend) Option { func WithBackend(backend Backend) Option {
queueCapacity := 0 queueCapacity := 0
queueCapacitySet := backend.QueueCapacity != nil queueCapacitySet := backend.QueueCapacity != nil

View File

@@ -60,7 +60,18 @@ func TestEngineRejectsRunBeforeCompletionWhenAdmissionIsFull(t *testing.T) {
awaitCapacitySignal(t, reader.entered, "first artifact read") awaitCapacitySignal(t, reader.entered, "first artifact read")
result, err := engine.Run(context.Background(), capacityInputRequest("http://second.example/v1")) canceledContext, cancel := context.WithCancel(context.Background())
cancel()
result, err := engine.Run(canceledContext, capacityInputRequest("http://canceled.example/v1"))
if result != nil || !errors.Is(err, context.Canceled) {
t.Fatalf("canceled capacity admission=(%+v, %v), want context cancellation", result, err)
}
var canceledCapacityErr *promptkit.CapacityError
if errors.Is(err, promptkit.ErrCapacityExceeded) || errors.As(err, &canceledCapacityErr) {
t.Fatalf("canceled admission exposed capacity rejection: %v", err)
}
result, err = engine.Run(context.Background(), capacityInputRequest("http://second.example/v1"))
if result != nil { if result != nil {
t.Fatalf("capacity rejection returned partial result: %+v", result) t.Fatalf("capacity rejection returned partial result: %+v", result)
} }
@@ -70,6 +81,21 @@ func TestEngineRejectsRunBeforeCompletionWhenAdmissionIsFull(t *testing.T) {
if errors.Is(err, promptkit.ErrInvalidRequest) || errors.Is(err, promptkit.ErrLLMGenerate) { if errors.Is(err, promptkit.ErrInvalidRequest) || errors.Is(err, promptkit.ErrLLMGenerate) {
t.Fatalf("capacity rejection had an unrelated category: %v", err) t.Fatalf("capacity rejection had an unrelated category: %v", err)
} }
var capacityErr *promptkit.CapacityError
if !errors.As(err, &capacityErr) || capacityErr == nil {
t.Fatalf("capacity rejection=%v, want CapacityError", err)
}
if capacityErr.BackendID != "limited" {
t.Fatalf("capacity backend ID=%q, want limited", capacityErr.BackendID)
}
capacityErr.BackendID = "changed"
result, err = engine.Run(context.Background(), capacityInputRequest("http://third.example/v1"))
var subsequentCapacityErr *promptkit.CapacityError
if result != nil || !errors.As(err, &subsequentCapacityErr) ||
subsequentCapacityErr == nil || subsequentCapacityErr.BackendID != "limited" {
t.Fatalf("subsequent capacity rejection=(%+v, %v), want independent limited CapacityError", result, err)
}
if calls := reader.callCount(); calls != 1 { if calls := reader.callCount(); calls != 1 {
t.Fatalf("artifact calls=%d, want only the admitted run", calls) t.Fatalf("artifact calls=%d, want only the admitted run", calls)
} }
@@ -174,6 +200,19 @@ func TestCapacityExceededSentinelContract(t *testing.T) {
if promptkit.ErrCapacityExceeded == nil { if promptkit.ErrCapacityExceeded == nil {
t.Fatal("ErrCapacityExceeded is nil") t.Fatal("ErrCapacityExceeded is nil")
} }
var nilCapacityErr *promptkit.CapacityError
zeroCapacityErr := &promptkit.CapacityError{}
populatedCapacityErr := &promptkit.CapacityError{BackendID: "limited"}
for _, capacityErr := range []error{nilCapacityErr, zeroCapacityErr} {
if !errors.Is(capacityErr, promptkit.ErrCapacityExceeded) {
t.Fatalf("capacity error=%v, want ErrCapacityExceeded", capacityErr)
}
}
var discoveredCapacityErr *promptkit.CapacityError
if !errors.As(populatedCapacityErr, &discoveredCapacityErr) || discoveredCapacityErr != populatedCapacityErr {
t.Fatalf("populated capacity error is not discoverable: %v", populatedCapacityErr)
}
for _, unrelated := range []error{ for _, unrelated := range []error{
promptkit.ErrInvalidConfig, promptkit.ErrInvalidConfig,
promptkit.ErrInvalidRequest, promptkit.ErrInvalidRequest,
@@ -181,7 +220,8 @@ func TestCapacityExceededSentinelContract(t *testing.T) {
promptkit.ErrValidation, promptkit.ErrValidation,
} { } {
if errors.Is(promptkit.ErrCapacityExceeded, unrelated) || if errors.Is(promptkit.ErrCapacityExceeded, unrelated) ||
errors.Is(unrelated, promptkit.ErrCapacityExceeded) { errors.Is(unrelated, promptkit.ErrCapacityExceeded) ||
errors.Is(populatedCapacityErr, unrelated) {
t.Fatalf("ErrCapacityExceeded aliases unrelated sentinel %v", unrelated) t.Fatalf("ErrCapacityExceeded aliases unrelated sentinel %v", unrelated)
} }
} }

40
capacity_error.go Normal file
View File

@@ -0,0 +1,40 @@
package promptkit
import (
"fmt"
"strings"
)
// CapacityError reports bounded admission rejected for a selected backend.
//
// Engine-produced values identify only rejection at Promptkit's bounded
// [Engine.Run] or [Engine.RunPrepared] admission boundary. BackendID is the
// normalized registered backend ID used for routing and capacity; endpoint
// overrides do not change it. Every engine-produced value is nonnil and has a
// nonblank BackendID. Provider errors, active-generation waiting, and caller
// cancellation are not represented by this type.
//
// Callers own returned values and may mutate BackendID without affecting engine
// state or another error. CapacityError and its default Go encoding have no
// stable JSON contract. Consumer-constructed values do not establish that an
// engine rejected work.
type CapacityError struct {
// BackendID is the normalized registered backend ID whose admission was
// rejected.
BackendID string
}
// Error returns diagnostic wording that is not a parsing contract. It is safe
// to call on a nil receiver or a value with a blank BackendID.
func (e *CapacityError) Error() string {
if e == nil || strings.TrimSpace(e.BackendID) == "" {
return ErrCapacityExceeded.Error()
}
return fmt.Sprintf("backend %q admission: %v", e.BackendID, ErrCapacityExceeded)
}
// Unwrap returns ErrCapacityExceeded so errors.Is and errors.As can be used
// together. It is safe to call on a nil receiver or a zero value.
func (e *CapacityError) Unwrap() error {
return ErrCapacityExceeded
}

View File

@@ -1,7 +1,9 @@
package promptkit package promptkit
import ( import (
"fmt"
"reflect" "reflect"
"unicode/utf8"
"gitea.maximumdirect.net/eric/promptkit/internal/domain" "gitea.maximumdirect.net/eric/promptkit/internal/domain"
"gitea.maximumdirect.net/eric/promptkit/internal/jsonvalue" "gitea.maximumdirect.net/eric/promptkit/internal/jsonvalue"
@@ -12,19 +14,52 @@ func toDomainRunRequest(req RunRequest) (domain.RunRequest, error) {
if err != nil { if err != nil {
return domain.RunRequest{}, err return domain.RunRequest{}, err
} }
appendedMessages, err := toDomainAppendedMessages(req.AppendedMessages)
if err != nil {
return domain.RunRequest{}, err
}
return domain.RunRequest{ return domain.RunRequest{
PromptID: req.PromptID, PromptID: req.PromptID,
PromptVersion: req.PromptVersion, PromptVersion: req.PromptVersion,
ProfileID: req.ProfileID, ProfileID: req.ProfileID,
SessionID: req.SessionID, SessionID: req.SessionID,
APIKey: req.APIKey, APIKey: req.APIKey,
Inputs: toDomainArtifactRefMap(req.Inputs), Inputs: toDomainArtifactRefMap(req.Inputs),
Vars: copyStringMap(req.Vars), Vars: copyStringMap(req.Vars),
Execution: execution, Execution: execution,
Validation: toDomainOutputContractPtr(req.Validation), Validation: toDomainOutputContractPtr(req.Validation),
AppendedMessages: appendedMessages,
}, nil }, nil
} }
func toDomainAppendedMessages(messages []RenderedMessage) ([]domain.RenderedMessage, error) {
if messages == nil {
return nil, nil
}
converted := make([]domain.RenderedMessage, len(messages))
for index, message := range messages {
if !utf8.ValidString(message.Content) {
return nil, fmt.Errorf("appended message %d content must be valid UTF-8", index)
}
role, err := domain.NormalizeMessageRole(message.Role)
if err != nil {
return nil, fmt.Errorf("appended message %d role: %w", index, err)
}
cacheControl, err := domain.NormalizeCacheControl(toDomainCacheControl(message.CacheControl))
if err != nil {
return nil, fmt.Errorf("appended message %d cache_control: %w", index, err)
}
converted[index] = domain.RenderedMessage{
Role: role,
Content: message.Content,
CacheControl: cacheControl,
}
}
return converted, nil
}
func fromDomainPreparedRun(prepared *domain.PreparedRun) *PreparedRun { func fromDomainPreparedRun(prepared *domain.PreparedRun) *PreparedRun {
if prepared == nil { if prepared == nil {
return nil return nil
@@ -170,6 +205,40 @@ func fromDomainExecutionTarget(target domain.ExecutionTarget) ExecutionTarget {
} }
} }
func fromDomainProfileInspection(inspection *domain.ProfileInspection) *ProfileInspection {
if inspection == nil {
return nil
}
return &ProfileInspection{
ProfileID: inspection.ProfileID,
EffectiveModelParams: fromDomainExecutionTarget(inspection.EffectiveModelParams),
APIKeyRequired: inspection.APIKeyRequired,
}
}
func fromDomainPromptInspection(inspection *domain.PromptInspection) *PromptInspection {
if inspection == nil {
return nil
}
inputs := make([]PromptInputDefinition, len(inspection.Inputs))
for i, input := range inspection.Inputs {
inputs[i] = PromptInputDefinition{
Name: input.Name,
Required: input.Required,
ContentType: input.ContentType,
Description: input.Description,
}
}
return &PromptInspection{
PromptID: inspection.PromptID,
PromptVersion: inspection.PromptVersion,
PromptHash: inspection.PromptHash,
DefaultProfileID: inspection.DefaultProfileID,
Inputs: inputs,
OutputContract: fromDomainOutputContract(inspection.OutputContract),
}
}
func fromDomainExecutionTargetPresence(presence domain.ExecutionTargetPresence) ExecutionTargetPresence { func fromDomainExecutionTargetPresence(presence domain.ExecutionTargetPresence) ExecutionTargetPresence {
return ExecutionTargetPresence{ return ExecutionTargetPresence{
Temperature: presence.Temperature, Temperature: presence.Temperature,
@@ -251,6 +320,16 @@ func fromDomainRenderedMessages(messages []domain.RenderedMessage) []RenderedMes
return out return out
} }
func toDomainCacheControl(cacheControl *CacheControl) *domain.CacheControl {
if cacheControl == nil {
return nil
}
return &domain.CacheControl{
Type: domain.CacheControlType(cacheControl.Type),
TTL: cacheControl.TTL,
}
}
func fromDomainCacheControl(cacheControl *domain.CacheControl) *CacheControl { func fromDomainCacheControl(cacheControl *domain.CacheControl) *CacheControl {
if cacheControl == nil { if cacheControl == nil {
return nil return nil

View File

@@ -0,0 +1,108 @@
package promptkit
import (
"reflect"
"strings"
"testing"
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
)
func TestToDomainAppendedMessages(t *testing.T) {
tests := []struct {
name string
messages []RenderedMessage
want []domain.RenderedMessage
wantErr string
privateRole string
privateText string
}{
{
name: "normalizes supported roles",
messages: []RenderedMessage{
{Role: " DeVeLoPeR ", Content: "developer"},
{Role: "SyStEm", Content: "system"},
{Role: "\tUsEr\n", Content: "user"},
{Role: "assistant", Content: "assistant"},
},
want: []domain.RenderedMessage{
{Role: domain.RoleDeveloper, Content: "developer"},
{Role: domain.RoleSystem, Content: "system"},
{Role: domain.RoleUser, Content: "user"},
{Role: domain.RoleAssistant, Content: "assistant"},
},
},
{
name: "invalid role UTF-8",
messages: []RenderedMessage{{Role: string([]byte{0xff}), Content: "private-content"}},
wantErr: "appended message 0 role",
},
{
name: "invalid content UTF-8",
messages: []RenderedMessage{{Role: "private-role", Content: string([]byte{0xff})}},
wantErr: "appended message 0 content",
},
{
name: "unsupported role",
messages: []RenderedMessage{{Role: "private-role", Content: "private-content"}},
wantErr: "appended message 0 role",
privateRole: "private-role",
privateText: "private-content",
},
{
name: "invalid cache control",
messages: []RenderedMessage{{Role: RoleUser, Content: "private-content", CacheControl: &CacheControl{Type: "private-cache"}}},
wantErr: "appended message 0 cache_control",
privateText: "private-content",
},
{
name: "empty and whitespace content",
messages: []RenderedMessage{
{Role: RoleUser, Content: ""},
{Role: RoleAssistant, Content: " \t\n "},
},
want: []domain.RenderedMessage{
{Role: domain.RoleUser, Content: ""},
{Role: domain.RoleAssistant, Content: " \t\n "},
},
},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
got, err := toDomainAppendedMessages(test.messages)
if test.wantErr != "" {
if err == nil || !strings.Contains(err.Error(), test.wantErr) {
t.Fatalf("error = %v, want %q", err, test.wantErr)
}
for _, privateValue := range []string{test.privateRole, test.privateText} {
if privateValue != "" && strings.Contains(err.Error(), privateValue) {
t.Fatalf("error exposed appended message data %q: %v", privateValue, err)
}
}
return
}
if err != nil {
t.Fatalf("toDomainAppendedMessages() error = %v", err)
}
if !reflect.DeepEqual(got, test.want) {
t.Fatalf("toDomainAppendedMessages() = %#v, want %#v", got, test.want)
}
})
}
}
func TestToDomainAppendedMessagesCopiesCacheControl(t *testing.T) {
cacheControl := &CacheControl{Type: CacheControlEphemeral, TTL: "1h"}
messages := []RenderedMessage{{Role: RoleUser, Content: "content", CacheControl: cacheControl}}
request, err := toDomainRunRequest(RunRequest{AppendedMessages: messages})
if err != nil {
t.Fatalf("toDomainRunRequest() error = %v", err)
}
messages[0].Content = "changed"
cacheControl.TTL = ""
if request.AppendedMessages[0].Content != "content" || request.AppendedMessages[0].CacheControl.TTL != "1h" {
t.Fatalf("domain request did not retain an independent appended-message copy: %#v", request.AppendedMessages)
}
}

47
doc.go
View File

@@ -3,23 +3,29 @@
// //
// Applications construct an [Engine] with [NewEngine], select filesystem or // Applications construct an [Engine] with [NewEngine], select filesystem or
// in-memory sources and optional engine-scoped [Backend] registrations, and // in-memory sources and optional engine-scoped [Backend] registrations, and
// call [Engine.Prepare] or [Engine.Run]. Concrete registries, repositories, // call [Engine.InspectPrompt], [Engine.InspectProfile], [Engine.Prepare],
// validators, and the built-in OpenAI-compatible client remain internal // [Engine.PrepareExecution], [Engine.Run], or [Engine.RunPrepared]. Concrete
// implementation details. // registries, repositories, validators, and the built-in OpenAI-compatible
// client remain internal implementation details.
// //
// # Concurrency and ownership // # Concurrency and ownership
// //
// An Engine supports concurrent Prepare and Run calls. Engine-local backend // An Engine supports concurrent InspectPrompt, InspectProfile, Prepare,
// policies bound admitted Run calls and model generations where configured, // PrepareExecution, Run, and RunPrepared calls. Engine-local backend policies
// while different backend pools and unlimited backends continue independently. // bound admitted Run and RunPrepared calls and model generations where
// An injected [LLMClient] or [ArtifactReader] can therefore still receive // configured, while different backend pools and unlimited backends continue
// concurrent calls and must be safe for that use. // independently. An injected [LLMClient] or [ArtifactReader] can therefore
// still receive concurrent calls and must be safe for that use.
// //
// NewEngine copies in-memory profiles and backend definitions. Prepare and Run // NewEngine copies in-memory profiles and backend definitions. Prepare,
// copy request maps, slices, pointer values, and JSON-compatible extra // PrepareExecution, and Run copy request maps, slices, pointer values, and
// parameters before using them. Returned values and values passed to extension // JSON-compatible extra parameters before using them. InspectPrompt and
// interfaces are likewise isolated from engine state. Callers own those copies // InspectProfile return copied inspection values. Returned values and values
// and may mutate them after the call that supplied or returned them. // passed to extension interfaces are likewise isolated from engine state.
// Callers own those copies and may mutate them after the call that supplied or
// returned them. [CapacityError] values are caller-owned and may be mutated
// without affecting engine state or another error. Immutable [GenerationError]
// values are also caller-owned and do not retain shared engine state.
// //
// # Security and sensitive data // # Security and sensitive data
// //
@@ -45,10 +51,17 @@
// [GenerateResponse], [ExecutionTargetPresence], and the string value types // [GenerateResponse], [ExecutionTargetPresence], and the string value types
// used by those values. // used by those values.
// //
// Construction values, including [Config], [Backend], [RunRequest], // Construction, inspection, handle, and error values, including [Config],
// [ArtifactRef], [ExecutionTargetOverride], [Profile], and // [Backend], [RunRequest], [ArtifactRef], [ExecutionTargetOverride], [Profile],
// [OpenAICompatibleProfileConfig], do not have stable JSON representations. // [OpenAICompatibleProfileConfig], [ProfileInspection],
// Direct API keys are nevertheless excluded from JSON for every public value. // [PromptInputDefinition], [PromptInspection], [PreparedExecution],
// [CapacityError], and [GenerationError], do not have stable JSON
// representations. Direct API keys are nevertheless excluded from JSON for
// every public value.
// Provider-derived [GenerationError] accessor values are untrusted and can
// contain sensitive request or schema fragments. Applications must apply their
// own disclosure policy before logging, displaying, or returning them.
// //
// JSON timestamps use time.Time's RFC 3339 encoding and are omitted when zero. // JSON timestamps use time.Time's RFC 3339 encoding and are omitted when zero.
// PreparedRun and RunResult durations are encoded as integer milliseconds in // PreparedRun and RunResult durations are encoded as integer milliseconds in

View File

@@ -40,11 +40,67 @@ validation, and default transport behavior. Source discovery, format
validation, and profile precedence are defined by the validation, and profile precedence are defined by the
[framework format reference](../formats.md). [framework format reference](../formats.md).
## Supply Embedded Application Defaults
Use `WithFallbackProfileFS` when an application packages profile definitions
that should apply unless an operator provides an ordinary configured profile
with the same ID. For example, an application can embed its defaults while
continuing to use `ProfileDir` for operator overrides:
```go
//go:embed profiles/*.yaml
var applicationProfiles embed.FS
engine, err := promptkit.NewEngine(promptkit.Config{
PromptDir: "prompts",
ProfileDir: operatorProfileDir,
},
promptkit.WithFallbackProfileFS(applicationProfiles, "profiles"),
)
```
Keep application-owned profile IDs and definitions in the embedded source.
Use the ordinary configured profile source for operator overrides. Leave
`operatorProfileDir` empty when the operator did not configure an override
directory; a non-empty path names an authoritative higher-precedence source,
so an unavailable or unreadable directory is an error rather than a reason to
fall back. The
[framework format reference](../formats.md#source-and-profile-precedence)
owns the exact profile format and lookup order; the
[`WithFallbackProfileFS` GoDoc](../../engine.go) owns its option contract and
validation rules.
## Inspect A Prompt Before Preparation
Use [`Engine.InspectPrompt`](../../engine.go) to check one configured prompt's
declared inputs and output workflow without creating placeholder inputs or
resolving a profile:
```go
inspection, err := engine.InspectPrompt(ctx, "meeting.summary", "")
if err != nil {
return err
}
for _, input := range inspection.Inputs {
// Compare the declared input with application configuration.
}
```
Use this configuration-time boundary when the application needs only the
declared prompt interface. Use `InspectProfile` separately when it must also
check a configured profile. Use `Prepare` when it needs inputs, schemas, or
rendered messages, and use prepared execution when that work must remain tied
to later execution. The method's [GoDoc](../../engine.go) owns exact fields,
hash, ownership, and error semantics.
## Prepare Without Model Execution ## Prepare Without Model Execution
[`Engine.Prepare`](../../engine.go) resolves the selected prompt and profile, [`Engine.Prepare`](../../engine.go) resolves the selected prompt and profile,
loads inputs and any structured-output schema, and renders messages without loads inputs and any structured-output schema, and renders messages without
calling a model client: calling a model client. Choose it when the prepared value is the final
inspection or persistence result and no later execution must be tied to that
exact snapshot:
```go ```go
prepared, err := engine.Prepare(ctx, promptkit.RunRequest{ prepared, err := engine.Prepare(ctx, promptkit.RunRequest{
@@ -61,12 +117,48 @@ shows a complete runnable setup with a prompt file, in-memory profile, and
inline input. Exact request requirements and prepared-result fields belong to inline input. Exact request requirements and prepared-result fields belong to
the [`RunRequest` and `PreparedRun` GoDoc](../../types.go). the [`RunRequest` and `PreparedRun` GoDoc](../../types.go).
## Prepare Now And Execute The Same Snapshot Later
Use [`Engine.PrepareExecution`](../../engine.go) when an application must
inspect or persist preflight details before deciding whether to start model
work, while ensuring that later execution uses those exact rendered messages,
inputs, target settings, and validation resources:
```go
preparedExecution, err := engine.PrepareExecution(ctx, promptkit.RunRequest{
PromptID: "meeting.summary",
Inputs: map[string]promptkit.ArtifactRef{
"note": promptkit.Inline("Synthetic meeting notes"),
},
})
if err != nil {
return err
}
defer preparedExecution.Discard()
details := preparedExecution.Details()
// Inspect or persist an application-selected safe subset of details.
result, err := engine.RunPrepared(ctx, preparedExecution)
```
Preparation does not call the model or reserve backend capacity.
`RunPrepared` executes from the retained snapshot rather than reloading
consumer sources. The handle is opaque in-process state, while `Details`
contains rendered content and remains subject to the application's data
handling policy. The
[`PreparedExecution` and method GoDoc](../../prepared_execution.go) and
[engine operation GoDoc](../../engine.go) own exact lifecycle, engine-binding,
credential, cancellation, timing, and error semantics.
## Execute And Validate ## Execute And Validate
[`Engine.Run`](../../engine.go) performs the same preparation, invokes the [`Engine.Run`](../../engine.go) performs the same preparation, invokes the
configured model client, classifies the generated artifact, and validates the configured model client, classifies the generated artifact, and validates the
content. A completed content check may return `ValidationFailed` in the result; content in one call. Choose it when the application does not need a preflight
an operational inability to validate returns an error. boundary tied to the eventual execution. A completed content check may return
`ValidationFailed` in the result; an operational inability to validate returns
an error.
The maintained The maintained
[offline execution example](../../examples/go-library/run/main.go) injects a [offline execution example](../../examples/go-library/run/main.go) injects a
@@ -81,6 +173,49 @@ semantics. The
[OpenAI-compatible integration contract](../integrations/openai-compatible-chat.md) [OpenAI-compatible integration contract](../integrations/openai-compatible-chat.md)
owns the built-in client's outbound HTTP behavior. owns the built-in client's outbound HTTP behavior.
### Repair A Structured Result
Set a small additional-call budget when a structurally invalid result can be
corrected automatically:
```go
request.Validation = &promptkit.OutputContract{
Format: promptkit.FormatJSON,
ValidationMode: promptkit.ValidationJSONSchema,
SchemaPath: "events.schema.json",
RepairAttempts: 1,
}
```
Each repair attempt is another model call, so it can increase latency and
usage; `RunResult.Usage` is cumulative and `Validation.RepairAttempts` reports
calls actually started. Exhaustion still returns the final failed validation
result. `basic` validation can also repair an empty candidate, but structural
validity is not evidence of factual or domain correctness. See the
[output-contract format reference](../formats.md#output-contract) and
[`OutputContract` GoDoc](../../types.go) for the exact budget and eligibility
rules.
### Append Already-Rendered Messages
An application can include an earlier assistant response and its own corrective
instruction in a fresh request without changing the configured prompt:
```go
request.AppendedMessages = []promptkit.RenderedMessage{
{Role: promptkit.RoleAssistant, Content: previousResponse},
{Role: promptkit.RoleUser, Content: correction},
}
result, err := engine.Run(ctx, request)
```
These messages are already rendered: Promptkit does not template or resolve
files in them, and they can contain sensitive model output or application
feedback. Promptkit remains stateless; every `Run` call re-resolves its current
sources and the application owns any semantic retry budget. When a
pre-execution equality check is required, use `PrepareExecution` and compare
its opaque rendered-prompt hash before invoking `RunPrepared`.
## Inputs, Profiles, And Overrides ## Inputs, Profiles, And Overrides
Use `File`, `Inline`, or `InlineWithURI` to construct request inputs. A request Use `File`, `Inline`, or `InlineWithURI` to construct request inputs. A request
@@ -90,13 +225,90 @@ replace execution settings or the complete output contract.
The [public value GoDoc](../../types.go) defines nil, empty, zero, replacement, The [public value GoDoc](../../types.go) defines nil, empty, zero, replacement,
copy, and credential behavior. The copy, and credential behavior. The
[framework format reference](../formats.md) defines how those request values [framework format reference](../formats.md) defines how those request values
interact with prompt definitions, file-backed profiles, built-ins, schemas, interact with prompt definitions, file-backed and application fallback
and framework defaults. profiles, built-ins, schemas, and framework defaults.
For programmatic profiles, For programmatic profiles,
[`OpenAICompatibleProfile`](../../profiles.go) converts ordinary [`OpenAICompatibleProfile`](../../profiles.go) converts ordinary
OpenAI-compatible settings into a value accepted by `WithProfiles`. OpenAI-compatible settings into a value accepted by `WithProfiles`.
### Alias A Built-In Profile
Give an application-owned profile ID a built-in base when prompts should select
the application ID while inheriting the built-in target. The child can override
only the setting it owns:
```go
promptkit.WithProfiles(promptkit.Profile{
ID: "weather-light",
BaseProfileID: "deepseek-4-flash",
ReasoningEffort: "high",
})
```
Select `weather-light` in a prompt or `RunRequest.ProfileID`; it remains the
reported selected profile. See the [profile inheritance format
reference](../formats.md#profile-inheritance) and the
[`Profile` GoDoc](../../types.go) for exact lookup, merging, and validation
behavior.
### Use The Rakestrawhome Built-In Profile
Set `RAKESTRAWHOME_INFERENCE_API_KEY` in the application environment, then
select `rakestrawhome-gemma-4-31b` as an ordinary profile ID. For example, a
prepared result identifies the selected built-in through
`BackendRakestrawHome`:
```go
prepared, err := engine.Prepare(ctx, promptkit.RunRequest{
PromptID: "meeting.summary",
ProfileID: "rakestrawhome-gemma-4-31b",
Inputs: inputs,
})
if err != nil {
return err
}
if prepared.SelectedBackendID != promptkit.BackendRakestrawHome {
return fmt.Errorf("unexpected backend %q", prepared.SelectedBackendID)
}
```
Do not register `rakestrawhome` manually. When adopting this built-in, remove
an existing `WithBackend` registration with that exact ID; retaining it causes
the intentional duplicate-ID configuration error. Direct request credentials
and runtime endpoint overrides remain supported under their ordinary GoDoc and
format contracts.
### Inspect A Profile Before Prompt Work
Use [`Engine.InspectProfile`](../../engine.go) to validate one configured
profile without constructing a synthetic prompt or placeholder inputs. It
resolves the profile's effective target but does not prepare or execute a
prompt:
```go
inspection, err := engine.InspectProfile(ctx, profileID)
if err != nil {
return err
}
target := inspection.EffectiveModelParams
if target.APIKeyEnv != "" {
// This is a configured optional environment lookup source.
} else if inspection.APIKeyRequired {
// Arrange a direct credential before later execution.
}
```
Use this configuration-time boundary when only the profile and its target need
checking. Use `Prepare` when the application also needs prompt, input, schema,
or rendering work; use prepared execution when that work must remain tied to a
later execution. A reported `APIKeyEnv` is a configured optional source, while
`APIKeyRequired` is the explicit local requirement. The
[credential format reference](../formats.md#credentials) and the method's
[GoDoc](../../engine.go) own the exact precedence, timing, result, and error
contracts.
### Set A Per-Run Session And Reasoning ### Set A Per-Run Session And Reasoning
Supply a direct session ID when one prompt should be correlated with a Supply a direct session ID when one prompt should be correlated with a
@@ -124,35 +336,86 @@ providers. The
[`RunRequest` and `ExecutionTargetOverride` GoDoc](../../types.go) owns the [`RunRequest` and `ExecutionTargetOverride` GoDoc](../../types.go) owns the
exact normalization, precedence, error, copying, and exposure contract. exact normalization, precedence, error, copying, and exposure contract.
### Register A Custom Backend ### Configure A Local OpenAI-Compatible Endpoint
Register a reusable OpenAI-compatible connection once, then select it from a Choose the smallest configuration that fits how the endpoint will be reused.
profile. This local backend limits model generation to two simultaneous calls;
because `QueueCapacity` is omitted, the engine admits up to 1024 additional #### Use An Endpoint-Only Profile
calls waiting behind them:
Put the endpoint directly on an in-memory profile when only that profile needs
it and shared backend identity or capacity policy is unnecessary:
```go ```go
engine, err := promptkit.NewEngine(promptkit.Config{ engine, err := promptkit.NewEngine(promptkit.Config{
PromptDir: "prompts", PromptDir: "prompts",
}, },
promptkit.WithBackend(promptkit.Backend{ promptkit.WithProfiles(promptkit.Profile{
ID: "local", ID: "local-summary",
Endpoint: "http://localhost:8000/v1", Endpoint: "http://localhost:8000/v1",
APIKeyEnv: "LOCAL_LLM_API_KEY", Model: "example-model",
ConcurrencyLimit: 2,
}), }),
)
```
Endpoint-only profiles have an empty backend ID and remain unrestricted by
backend capacity policy.
#### Use The Conventional Local Backend
Use `LocalBackend` when profiles should share the conventional `local`
identity, endpoint, and concurrency limit:
```go
engine, err := promptkit.NewEngine(promptkit.Config{
PromptDir: "prompts",
},
promptkit.WithBackend(
promptkit.LocalBackend("http://localhost:8000/v1", 2),
),
promptkit.WithProfiles(promptkit.Profile{ promptkit.WithProfiles(promptkit.Profile{
ID: "local-summary", ID: "local-summary",
BackendID: "local", BackendID: promptkit.BackendLocal,
Model: "example-model", Model: "example-model",
}), }),
) )
``` ```
The helper is explicit: it does not pre-register a backend or read environment
variables. Supplying a positive limit leaves queue capacity omitted, so normal
backend registration selects the existing default waiting capacity of 1024.
The returned value still enters the engine through `WithBackend`.
#### Configure A Complete Backend
Use a keyed `Backend` value for authentication, extra request parameters, an
explicit queue capacity, a custom ID, or multiple local endpoints:
```go
noWaiting := 0
engine, err := promptkit.NewEngine(promptkit.Config{
PromptDir: "prompts",
},
promptkit.WithBackend(promptkit.Backend{
ID: "local-gpu",
Endpoint: "http://gpu-host:8000/v1",
APIKeyEnv: "LOCAL_GPU_API_KEY",
ExtraParams: map[string]any{"provider_option": "enabled"},
ConcurrencyLimit: 2,
QueueCapacity: &noWaiting,
}),
promptkit.WithProfiles(promptkit.Profile{
ID: "gpu-summary",
BackendID: "local-gpu",
Model: "example-model",
}),
)
```
Use distinct custom IDs when registering multiple local endpoints.
Registrations belong to one engine and custom IDs cannot replace built-ins. Registrations belong to one engine and custom IDs cannot replace built-ins.
The [`Backend` and `WithBackend` GoDoc](../../backends.go) defines validation, The [`Backend`, `LocalBackend`, and `WithBackend` GoDoc](../../backends.go)
copying, uniqueness, exact concurrency-field semantics, and request-default defines exact construction, validation, copying, uniqueness, concurrency, and
behavior. request-default behavior.
Both file-backed and in-memory profiles select a registration through Both file-backed and in-memory profiles select a registration through
`backend` or `Profile.BackendID`. Profile and request endpoint overrides retain `backend` or `Profile.BackendID`. Profile and request endpoint overrides retain
@@ -173,8 +436,8 @@ zero:
```go ```go
noWaiting := 0 noWaiting := 0
backend := promptkit.Backend{ backend := promptkit.Backend{
ID: "local", ID: "local-gpu",
Endpoint: "http://localhost:8000/v1", Endpoint: "http://gpu-host:8000/v1",
ConcurrencyLimit: 2, ConcurrencyLimit: 2,
QueueCapacity: &noWaiting, QueueCapacity: &noWaiting,
} }
@@ -236,6 +499,11 @@ When a limited backend has admitted all active and waiting calls, handle
```go ```go
result, err := engine.Run(ctx, request) result, err := engine.Run(ctx, request)
if errors.Is(err, promptkit.ErrCapacityExceeded) { if errors.Is(err, promptkit.ErrCapacityExceeded) {
var capacityErr *promptkit.CapacityError
if errors.As(err, &capacityErr) {
// Record capacityErr.BackendID using application-owned diagnostics.
}
// Apply application policy: shed work, report overload, or retry later. // Apply application policy: shed work, report overload, or retry later.
} }
``` ```
@@ -243,8 +511,27 @@ if errors.Is(err, promptkit.ErrCapacityExceeded) {
A rejected call returns no partial result and does not invoke the model A rejected call returns no partial result and does not invoke the model
client. Promptkit does not prescribe retries or map this error to an HTTP client. Promptkit does not prescribe retries or map this error to an HTTP
status; those choices remain with the consuming application. The status; those choices remain with the consuming application. The
[`Engine.Run` and error GoDoc](../../engine.go) owns exact error and [`CapacityError` GoDoc](../../capacity_error.go) owns the exact typed-error
cancellation identities. contract, while the [`Engine.Run` and error GoDoc](../../engine.go) owns broad
error and cancellation identities.
For a non-2xx response from the built-in OpenAI-compatible client, inspect the
status and deliberately selected provider diagnostic when useful:
```go
var generationErr *promptkit.GenerationError
if errors.As(err, &generationErr) {
status := generationErr.StatusCode()
message := generationErr.ProviderMessage()
_, _ = status, message // Apply application retry and presentation policy.
}
```
All provider fields are untrusted and can contain sensitive request or schema
fragments. Do not log, display, or return them without an application-specific
disclosure policy. Promptkit does not assign retry or presentation behavior.
The [`GenerationError` GoDoc](../../generation_error.go) owns the exact typed
error contract.
## Application Boundary ## Application Boundary

View File

@@ -34,6 +34,7 @@ Start with:
| Root public API | The [architecture policy](policy/architecture.md), [consumer guide](consumers/pkg-promptkit.md), [testing policy](policy/testing.md), and existing GoDoc. | | Root public API | The [architecture policy](policy/architecture.md), [consumer guide](consumers/pkg-promptkit.md), [testing policy](policy/testing.md), and existing GoDoc. |
| Prompt, profile, or schema formats | The [framework format reference](formats.md), owning parser or validator package, and [documentation policy](policy/documentation.md). | | Prompt, profile, or schema formats | The [framework format reference](formats.md), owning parser or validator package, and [documentation policy](policy/documentation.md). |
| Source loading or validation | The [framework format reference](formats.md), [internal source document](internal/sources.md), and owning package tests. | | Source loading or validation | The [framework format reference](formats.md), [internal source document](internal/sources.md), and owning package tests. |
| Maintained external catalogs or built-ins | The [internal source document](internal/sources.md), [framework format reference](formats.md), [testing policy](policy/testing.md), both catalog module repositories, and each catalog's release procedure. |
| Model-client behavior | The [OpenAI-compatible integration contract](integrations/openai-compatible-chat.md), [internal model-client document](internal/llm.md), and owning package tests. | | Model-client behavior | The [OpenAI-compatible integration contract](integrations/openai-compatible-chat.md), [internal model-client document](internal/llm.md), and owning package tests. |
| Internal package implementation | The [architecture policy](policy/architecture.md), [internal component overview](internal/overview.md), and focused internal document listed for that package. | | Internal package implementation | The [architecture policy](policy/architecture.md), [internal component overview](internal/overview.md), and focused internal document listed for that package. |
| Tests or test fixtures | The [testing policy](policy/testing.md), owning package, and focused internal document listed by the component overview. | | Tests or test fixtures | The [testing policy](policy/testing.md), owning package, and focused internal document listed by the component overview. |
@@ -45,11 +46,17 @@ For cross-cutting changes, follow every applicable row. Do not create
placeholder documents for packages, APIs, or integrations that do not yet placeholder documents for packages, APIs, or integrations that do not yet
exist. exist.
## Maintainer-Run Validation ## Maintainer Validation
Promptkit does not currently use hosted CI. Maintainers are responsible for This section is the canonical local validation workflow for Promptkit. Run
running the documented checks before accepting changes. Run the default Go every command from the repository root before accepting a change. The test
validation from the Promptkit repository root: suite and maintained examples are deterministic, offline, and require no real
provider credentials.
### Tests, Analysis, Build, And Examples
Run the ordinary and race-enabled suites, static analysis, the build, and both
maintained consumer examples:
```sh ```sh
go test ./... go test ./...
@@ -57,85 +64,171 @@ go test -race ./...
go vet ./... go vet ./...
go build ./... go build ./...
go run ./examples/go-library/prepare go run ./examples/go-library/prepare
go run ./examples/go-library/run
``` ```
Check formatting across every tracked Go file: Both examples must exit successfully. Review their JSON output: preparation
must report the selected offline prompt, profile, model, and message count;
execution must report the deterministic generated output, successful
validation, selected offline model, and usage. Neither command may contact a
provider or require credentials.
### Go Formatting
Check every tracked Go file. The final command must succeed and the captured
list must be empty:
```sh ```sh
gofmt -l $(git ls-files '*.go') unformatted=$(
git ls-files '*.go' |
while IFS= read -r go_file
do
gofmt -l "$go_file"
done
)
test -z "$unformatted"
``` ```
The formatting command must produce no paths. Follow every added or changed ### Local Markdown Links
Markdown link and confirm its target exists. Finally, check whitespace:
Use the Python standard library to verify every repository-relative Markdown
target and local heading fragment. The check is offline and prints nothing on
success:
```sh
python3 - <<'PY'
from pathlib import Path
import re
import subprocess
import sys
from urllib.parse import unquote
root = Path.cwd().resolve()
markdown_files = [
root / name
for name in subprocess.check_output(
["git", "ls-files", "*.md"], text=True
).splitlines()
]
link_pattern = re.compile(r"!?\[[^]]*\]\(([^)]+)\)")
heading_pattern = re.compile(r"^#{1,6}\s+(.+?)\s*#*\s*$")
scheme_pattern = re.compile(r"^[a-z][a-z0-9+.-]*:", re.IGNORECASE)
def markdown_lines(path):
in_fence = False
fence = ""
for line in path.read_text(encoding="utf-8").splitlines():
stripped = line.lstrip()
marker = stripped[:3]
if marker in {"```", "~~~"}:
if not in_fence:
in_fence = True
fence = marker
elif marker == fence:
in_fence = False
fence = ""
continue
if not in_fence:
yield line
anchor_cache = {}
def anchors(path):
if path in anchor_cache:
return anchor_cache[path]
found = set()
counts = {}
for line in markdown_lines(path):
match = heading_pattern.match(line)
if not match:
continue
heading = re.sub(r"<[^>]+>", "", match.group(1)).replace("`", "")
base = re.sub(r"[^\w\- ]", "", heading.lower()).replace(" ", "-")
count = counts.get(base, 0)
counts[base] = count + 1
found.add(base if count == 0 else f"{base}-{count}")
anchor_cache[path] = found
return found
failures = []
for source in markdown_files:
text = "\n".join(markdown_lines(source))
for match in link_pattern.finditer(text):
target = match.group(1).strip()
if target.startswith("<") and target.endswith(">"):
target = target[1:-1]
if scheme_pattern.match(target) or target.startswith("//"):
continue
path_text, separator, fragment = target.partition("#")
destination = source if not path_text else source.parent / unquote(path_text)
try:
destination = destination.resolve()
destination.relative_to(root)
except ValueError:
failures.append(f"{source.relative_to(root)}: escapes repository: {target}")
continue
if not destination.exists():
failures.append(f"{source.relative_to(root)}: missing target: {target}")
continue
if separator and destination.suffix.lower() == ".md":
fragment = unquote(fragment).lower()
if fragment not in anchors(destination):
failures.append(f"{source.relative_to(root)}: missing anchor: {target}")
if failures:
print("\n".join(failures), file=sys.stderr)
raise SystemExit(1)
PY
```
### Repository Hygiene And Review
Reject an active Go workspace, tracked workspace files, a vendor tree, or a
module replacement:
```sh
case "$(go env GOWORK)" in
''|off) ;;
*) printf '%s\n' 'an active Go workspace is not allowed' >&2; exit 1 ;;
esac
test -z "$(git ls-files go.work go.work.sum)"
test ! -e vendor
if grep -Eq '^[[:space:]]*replace([[:space:]]|\()' go.mod
then
printf '%s\n' 'go.mod contains a replacement' >&2
exit 1
fi
```
Check whitespace in both unstaged and staged changes. List ignored files and
scan tracked content for common credential forms:
```sh ```sh
git diff --check git diff --check
git diff --cached --check
test -z "$(git ls-files --others --ignored --exclude-standard)"
credential_pattern='-----BEGIN ([A-Z0-9]+ )?PRIV''ATE KEY-----|AKI''A[0-9A-Z]{16}|gh[pousr]_[A-Za-z0-9]{36,}|sk-[A-Za-z0-9]{32,}'
if git grep -nEI -e "$credential_pattern" -- .
then
printf '%s\n' 'possible credential found' >&2
exit 1
fi
``` ```
Documentation-only work does not require unrelated new tests, but it still Inspect `git status --short --untracked-files=all` and the complete diff before
requires link validation and `git diff --check`. Run the Go validation whenever accepting a change. The status may contain only the intended source changes
documentation changes commands, examples, generated output, or another during development. Reject credentials, private keys, environment files,
behavior checked by the module. generated binaries, test or coverage output, downloaded assets, template
residue, and any other artifact that does not belong in source control. The
credential scan catches common forms but does not replace inspection of the
actual change.
## Focused Validation After committing the accepted change, require a clean candidate:
Use focused checks while iterating, then run the complete validation sequence
before accepting the change. The root package supports:
```sh ```sh
go test . test -z "$(git status --porcelain)"
go vet .
go build .
``` ```
Filter tests by name without assuming a fixed internal package layout:
```sh
go test ./... -run 'TestName'
```
Replace `TestName` with a useful regular expression. Target only paths that
exist, and consult the internal component overview for their owning
documentation. A filtered or package-specific run does not replace the
complete repository validation.
## Coordinated Work With Scriptorium
Promptkit and Scriptorium must remain independently valid. For temporary local
integration, use either a Go workspace outside both repositories or an
uncommitted replacement in the consuming module.
If the repositories are sibling directories, run the workspace commands from
their parent directory:
```sh
go work init ./promptkit ./scriptorium
go work sync
```
Use the workspace only for coordinated local checks. From the same parent
directory, remove it when finished:
```sh
rm -f go.work go.work.sum
```
Alternatively, from the Scriptorium repository root, temporarily point its
Promptkit dependency at the sibling checkout:
```sh
go mod edit -replace gitea.maximumdirect.net/eric/promptkit=../promptkit
```
After coordinated checks, remove the replacement and reconcile module
metadata:
```sh
go mod edit -dropreplace gitea.maximumdirect.net/eric/promptkit
go mod tidy
```
Never commit `go.work`, `go.work.sum`, or a local filesystem `replace`
directive. Before committing in either repository, inspect its module files and
working tree independently. Published consumer versions must depend on a tagged
Promptkit version, not a workspace, local replacement, or unpublished commit.

View File

@@ -9,9 +9,10 @@ explains how to select these sources and invoke the engine. The
owns the resulting outbound wire behavior. owns the resulting outbound wire behavior.
Prompt and profile sources recursively discover files ending in `.yaml` or Prompt and profile sources recursively discover files ending in `.yaml` or
`.yml`. YAML decoding is strict: unknown fields are errors for the selected `.yml`. Each prompt-definition and profile file contains exactly one YAML
definition. Definitions are selected by their YAML `id`, not their file name document; comments and trailing whitespace are allowed. YAML decoding is
or directory. strict: unknown fields are errors for the selected definition. Definitions are
selected by their YAML `id`, not their file name or directory.
## Prompt Definitions ## Prompt Definitions
@@ -58,6 +59,11 @@ When a request omits a version, the selected prompt ID must identify exactly
one definition. When it supplies a version, the ID and version pair must be one definition. When it supplies a version, the ID and version pair must be
unique. unique.
Exact prompt inspection uses this same configured source, strict decoding,
referenced content-file resolution, and ID/version selection. It reports the
selected definition's declared metadata without changing the prompt format or
executing the definition.
### Inputs ### Inputs
Each `inputs` item has these fields: Each `inputs` item has these fields:
@@ -76,14 +82,27 @@ allowed.
### Messages And Templates ### Messages And Templates
Each message has a non-empty `role` and exactly one of: Each message has a `role` that Promptkit trims and lowercases. It must then be
exactly one of `developer`, `system`, `user`, or `assistant`; blank, custom,
`tool`, and `function` roles are invalid. This intentionally tightens the
previous nonblank-string rule. Consumers migrating to the next minor release
must update any nonstandard prompt-definition roles before upgrading.
Each message also has exactly one of:
- `content`, containing an inline Go template; or - `content`, containing an inline Go template; or
- `content_file`, naming a file whose contents are the Go template. - `content_file`, naming a file whose contents are the Go template.
For directory and `fs.FS` prompt sources, `content_file` resolves relative to `content_file` must be a relative path. It resolves from the directory that
the prompt file and remains within the source root. `WithPromptFile` also contains the prompt file and must remain within the configured prompt source
resolves it relative to that file. root; parent components are allowed only when the resolved target remains
inside that root. Absolute paths and paths that escape the root are rejected.
Operating-system directory and single-file sources also reject symlink targets
outside the root, while injected `fs.FS` sources apply containment in that
filesystem's relative path namespace. For `WithPromptFile`, the source root is
the directory containing the selected prompt file. Promptkit uses the parsed
path text exactly after checking separately that it is not blank, so leading
and trailing whitespace can name real filesystem entries.
Request variables are the template data, so a variable named `audience` is Request variables are the template data, so a variable named `audience` is
referenced as `{{.audience}}`. The `{{input "note"}}` helper renders the body referenced as `{{.audience}}`. The `{{input "note"}}` helper renders the body
@@ -113,7 +132,7 @@ outbound integration determines its wire representation.
| `format` | yes | `text`, `markdown`, or `json`. | | `format` | yes | `text`, `markdown`, or `json`. |
| `validation_mode` | yes | `none`, `basic`, `json`, or `json_schema`. | | `validation_mode` | yes | `none`, `basic`, `json`, or `json_schema`. |
| `schema_path` | for `json_schema` | Path to a schema in the configured schema source. | | `schema_path` | for `json_schema` | Path to a schema in the configured schema source. |
| `repair_attempts` | no | Integer zero or greater; omitted means zero. | | `repair_attempts` | no | Integer from zero through three; omitted means zero. A positive value requires `basic`, `json`, or `json_schema` validation. |
The validation modes behave as follows: The validation modes behave as follows:
@@ -123,14 +142,31 @@ The validation modes behave as follows:
- `json_schema` requires valid JSON that satisfies the selected schema. - `json_schema` requires valid JSON that satisfies the selected schema.
`format` controls output artifact metadata. JSON Schema mode also supplies the `format` controls output artifact metadata. JSON Schema mode also supplies the
schema to compatible model clients as structured-output metadata. The public schema to compatible model clients as structured-output metadata. Plain `json`
engine does not install an output repairer, so its validation is single-pass validation accepts every valid JSON value and does not request a provider-native
even when a positive `repair_attempts` value is present. JSON-object constraint.
`repair_attempts` counts additional generation calls after a failed validation.
Zero is single-pass. With a positive eligible budget, Promptkit stops at the
first valid candidate. If the budget is exhausted, it returns the final
candidate and its complete failed validation result; generation and operational
validation failures remain errors. `none` never permits repair.
A request-level `OutputContract` replaces the complete prompt output contract. A request-level `OutputContract` replaces the complete prompt output contract.
It does not merge individual fields. If its format is empty, Promptkit uses It does not merge individual fields. If its format is empty, Promptkit uses
`text`. `text`.
## Built-In Backends
Every engine provides these reserved OpenAI-compatible backend IDs. Consumers
must not register either ID with `WithBackend`; exact registration and
reservation behavior belongs to the [`Backend` GoDoc](../backends.go).
| ID | Base endpoint | API-key environment variable | Active generation limit | Default queue capacity |
| --- | --- | --- | ---: | ---: |
| `openrouter` | `https://openrouter.ai/api/v1` | `OPENROUTER_API_KEY` | 16 | 1024 |
| `rakestrawhome` | `https://inference.ai.rakestrawhome.com/v1` | `RAKESTRAWHOME_INFERENCE_API_KEY` | 4 | 1024 |
## Profile Definitions ## Profile Definitions
A profile supplies model execution settings: A profile supplies model execution settings:
@@ -149,11 +185,21 @@ extra_params:
provider_option: enabled provider_option: enabled
``` ```
A derived profile can use a named base and override only the settings it owns:
```yaml
id: local-summary-fast
base_profile: local-summary
timeout_seconds: 30
reasoning_effort: low
```
| Field | Required | Meaning | | Field | Required | Meaning |
| --- | --- | --- | | --- | --- | --- |
| `id` | yes | Non-empty profile identifier. IDs must be unique within one source. | | `id` | yes | Profile identifier, trimmed before selection and publication. It must be non-empty after trimming and unique within one source after normalization. |
| `backend` | unless `endpoint` is present | Backend registry ID. It is trimmed and registry membership is checked when the profile is prepared. | | `base_profile` | no | One optional parent profile ID. A derived profile may inherit target fields from it. |
| `endpoint` | unless `backend` is present | Non-empty OpenAI-compatible base URL, including an API version path when required. When both connection fields are present, this overrides the backend endpoint without changing backend identity. | | `backend` | unless `endpoint` is present | Backend registry ID. It is trimmed and registry membership is checked when the profile is prepared or inspected. |
| `endpoint` | unless `backend` is present | OpenAI-compatible base URL, including an API version path when required. A nonempty value is trimmed and must be absolute HTTP or HTTPS with a host and without user information, a query, or a fragment. When both connection fields are present, this overrides the backend endpoint without changing backend identity. |
| `model` | yes | Non-empty provider model name. | | `model` | yes | Non-empty provider model name. |
| `temperature` | no | Number from 0 through 2. | | `temperature` | no | Number from 0 through 2. |
| `max_tokens` | no | Integer zero or greater. | | `max_tokens` | no | Integer zero or greater. |
@@ -161,16 +207,21 @@ extra_params:
| `timeout_seconds` | no | Per-generation deadline in whole seconds; integer zero or greater. | | `timeout_seconds` | no | Per-generation deadline in whole seconds; integer zero or greater. |
| `service_tier` | no | Provider-specific request tier. | | `service_tier` | no | Provider-specific request tier. |
| `reasoning_effort` | no | Provider-specific reasoning setting. | | `reasoning_effort` | no | Provider-specific reasoning setting. |
| `api_key_env` | no | Name of an environment variable containing the API key. | | `api_key_env` | no | Optional environment-variable lookup source for an API key. |
| `extra_params` | no | JSON-compatible provider-specific outbound fields. | | `extra_params` | no | JSON-compatible provider-specific outbound fields. |
Raw `api_key` is prohibited in profile YAML. Store only an environment Raw `api_key` is prohibited in profile YAML. Store only an environment
variable name in `api_key_env`. variable name in `api_key_env`.
A standalone profile must provide a model and at least one of `backend` or
`endpoint`. A derived profile may omit those target fields because its selected
base chain can provide them. Local parsing still validates a derived profile's
own ID, supplied endpoint, execution-setting bounds, and `extra_params`.
Promptkit does not infer a backend from a model or endpoint. Endpoint-only Promptkit does not infer a backend from a model or endpoint. Endpoint-only
profiles remain supported and have no effective backend ID. profiles remain supported and have no effective backend ID.
The engine always provides the built-in `openrouter` ID. Consumers can add The engine always provides the built-in `openrouter` and `rakestrawhome` IDs.
engine-scoped IDs with Consumers can add engine-scoped IDs with
[`WithBackend`](../backends.go); exact registration validation belongs to its [`WithBackend`](../backends.go); exact registration validation belongs to its
GoDoc. GoDoc.
@@ -178,30 +229,33 @@ GoDoc.
objects with string keys. Keys must be non-empty. With the built-in client, objects with string keys. Keys must be non-empty. With the built-in client,
they also cannot collide with the standard fields listed in the they also cannot collide with the standard fields listed in the
[outbound request contract](integrations/openai-compatible-chat.md#request-body). [outbound request contract](integrations/openai-compatible-chat.md#request-body).
Excessively deep or large JSON-shaped values are rejected for safety.
### Defaults And Overrides ### Defaults And Overrides
Execution settings resolve in this order: Execution settings resolve in this order:
1. framework defaults; 1. the framework timeout baseline;
2. the selected backend, when the profile names one; 2. the selected backend, when the profile names one;
3. the selected profile; and 3. the selected profile; and
4. request `ExecutionTargetOverride` values. 4. request `ExecutionTargetOverride` values.
The framework defaults are: The framework baseline is:
| Setting | Default | | Setting | Default |
| --- | --- | | --- | --- |
| `temperature` | `0` | | `temperature` | Unspecified and omitted from compatible provider requests unless a profile or runtime override selects it. |
| `max_tokens` | `0` | | `max_tokens` | Unspecified and omitted from compatible provider requests unless a profile or runtime override selects it. |
| `top_p` | `1` | | `top_p` | Unspecified and omitted from compatible provider requests unless a profile or runtime override selects it. |
| `timeout_seconds` | `600` | | `timeout_seconds` | `600` |
Numeric zero in a file or in-memory profile means that the profile does not Numeric zero in a file or in-memory profile does not select a numeric value.
replace the framework default. Numeric request overrides use pointers, so an For `temperature`, `max_tokens`, and `top_p`, it leaves the provider control
explicit zero is preserved. In particular, an explicit request unspecified. For `timeout_seconds`, it retains the framework deadline. Numeric
`timeout_seconds` of zero disables the per-generation deadline while leaving request overrides use pointers, so an explicit zero is retained and sent to
the caller context and transport timeout intact. compatible providers. In particular, an explicit request `timeout_seconds` of
zero disables the per-generation deadline while leaving the caller context and
transport timeout intact.
Non-empty profile strings replace backend defaults, and non-empty request Non-empty profile strings replace backend defaults, and non-empty request
strings replace both. Request reasoning is the exception: a nil strings replace both. Request reasoning is the exception: a nil
@@ -219,54 +273,86 @@ defines how the effective settings are serialized.
### Source And Profile Precedence ### Source And Profile Precedence
An explicit request profile ID takes precedence over the prompt's An explicit request profile ID takes precedence over the prompt's
`default_profile`. If neither is present, preparation fails. `default_profile`. If neither is present, preparation fails. Exact profile
inspection instead takes one explicit profile ID and does not use a prompt
default.
Profile sources resolve matching IDs in this order: Profile sources resolve matching IDs in this order:
1. in-memory profiles supplied with `WithProfiles`; 1. in-memory profiles supplied with `WithProfiles`;
2. a profile file, `fs.FS`, or configured profile directory; and 2. the ordinary configured source selected by a profile file, `fs.FS`, or
3. embedded built-in profiles. configured profile directory;
3. application fallback profiles supplied with `WithFallbackProfileFS`; and
4. maintained external catalog profiles.
A higher-precedence source falls back only when the profile is absent. An A profile source supplies a complete definition; definitions and their fields
invalid matching profile is an error and does not fall back. In-memory are not merged across sources. A higher-precedence source falls back only when
`Profile` values follow the same ranges as YAML profiles. They use the requested profile ID is absent. An invalid matching profile is an error and
`APIKeyRequired` for request-scoped credentials instead of `api_key_env`. does not fall back. In-memory `Profile` values follow the same ranges as YAML
profiles. They use `APIKeyRequired` for request-scoped credentials instead of
`api_key_env`. Preparation and exact profile inspection use this same source
precedence.
When a selected definition names `base_profile`, every profile ID in that
chain is looked up through this same precedence order. A higher-precedence
definition therefore shadows a lower-precedence definition of the same base
ID, including a built-in. References are not source-qualified.
### Profile Inheritance
Promptkit resolves one linear base chain of at most 32 profiles, including the
selected profile. It merges settings from the root base to the selected leaf.
The leaf's `id` remains the selected profile identity. Nonblank string fields
(`backend`, `endpoint`, `model`, `service_tier`, `reasoning_effort`, and
`api_key_env`) and nonzero numeric fields replace inherited values. A nonempty
`extra_params` map replaces the complete inherited map rather than merging
keys, and `APIKeyRequired: true` remains true through the chain. Backend and
endpoint are independent: replacing one does not clear the other.
There is no profile-level clearing syntax. Blank strings, zero numbers, false,
and empty maps remain unspecified and inherit from a base. Use existing
presence-aware request overrides where an execution needs an explicit zero or
empty reasoning setting.
An absent directly selected profile reports the ordinary not-found error. Once
the selected profile exists, a missing base, cycle, overlong chain, or
incomplete resolved target is a profile-load failure. Ordinary operations
resolve chains afresh; prepared execution retains the fully resolved target.
## Built-In Profile Catalog ## Built-In Profile Catalog
Every built-in selects the `openrouter` backend. The engine's built-in backend Every built-in profile selects one maintained built-in backend and inherits
registry supplies `https://openrouter.ai/api/v1` and the environment-variable that backend's connection and credential metadata. Profile files do not repeat
name `OPENROUTER_API_KEY`, so individual profiles contain only model and those values. A configured, application fallback, or in-memory profile with
generation settings. Built-in profile files do not repeat those connection the same profile ID takes precedence.
values. A custom or in-memory profile with the same profile ID takes
precedence.
| Provider | ID | Model | | Provider | ID | Backend | Model |
| --- | --- | --- | | --- | --- | --- | --- |
| aion-labs | `aion-2` | `aion-labs/aion-2.0` | | aion-labs | `aion-2` | `openrouter` | `aion-labs/aion-2.0` |
| anthropic | `claude-fable-latest` | `~anthropic/claude-fable-latest` | | anthropic | `claude-fable-latest` | `openrouter` | `~anthropic/claude-fable-latest` |
| anthropic | `claude-haiku-latest` | `~anthropic/claude-haiku-latest` | | anthropic | `claude-haiku-latest` | `openrouter` | `~anthropic/claude-haiku-latest` |
| anthropic | `claude-opus-latest` | `~anthropic/claude-opus-latest` | | anthropic | `claude-opus-latest` | `openrouter` | `~anthropic/claude-opus-latest` |
| anthropic | `claude-sonnet-latest` | `~anthropic/claude-sonnet-latest` | | anthropic | `claude-sonnet-latest` | `openrouter` | `~anthropic/claude-sonnet-latest` |
| deepseek | `deepseek-3-2` | `deepseek/deepseek-v3.2` | | deepseek | `deepseek-3-2` | `openrouter` | `deepseek/deepseek-v3.2` |
| deepseek | `deepseek-4-flash` | `deepseek/deepseek-v4-flash` | | deepseek | `deepseek-4-flash` | `openrouter` | `deepseek/deepseek-v4-flash` |
| deepseek | `deepseek-4-pro` | `deepseek/deepseek-v4-pro` | | deepseek | `deepseek-4-pro` | `openrouter` | `deepseek/deepseek-v4-pro` |
| google | `gemini-2-flash` | `google/gemini-2.5-flash` | | google | `gemini-2-flash` | `openrouter` | `google/gemini-2.5-flash` |
| google | `gemini-2-flash-lite` | `google/gemini-2.5-flash-lite` | | google | `gemini-2-flash-lite` | `openrouter` | `google/gemini-2.5-flash-lite` |
| google | `gemini-2-pro` | `google/gemini-2.5-pro` | | google | `gemini-2-pro` | `openrouter` | `google/gemini-2.5-pro` |
| google | `gemini-3-flash-lite` | `google/gemini-3.1-flash-lite` | | google | `gemini-3-flash-lite` | `openrouter` | `google/gemini-3.1-flash-lite` |
| google | `gemini-flash-latest` | `~google/gemini-flash-latest` | | google | `gemini-flash-latest` | `openrouter` | `~google/gemini-flash-latest` |
| google | `gemini-pro-latest` | `~google/gemini-pro-latest` | | google | `gemini-pro-latest` | `openrouter` | `~google/gemini-pro-latest` |
| google | `gemma-4-31b` | `google/gemma-4-31b-it:exacto` | | google | `gemma-4-31b` | `openrouter` | `google/gemma-4-31b-it:exacto` |
| minimax | `minimax-m2` | `minimax/minimax-m2.5` | | google | `rakestrawhome-gemma-4-31b` | `rakestrawhome` | `google/gemma-4-31b-it` |
| minimax | `minimax-m3` | `minimax/minimax-m3` | | minimax | `minimax-m2` | `openrouter` | `minimax/minimax-m2.5` |
| mistral | `mistral-large-2512` | `mistralai/mistral-large-2512` | | minimax | `minimax-m3` | `openrouter` | `minimax/minimax-m3` |
| mistral | `mistral-medium-3-5` | `mistralai/mistral-medium-3-5` | | mistral | `mistral-large-2512` | `openrouter` | `mistralai/mistral-large-2512` |
| mistral | `mistral-small-3` | `mistralai/mistral-small-3.2-24b-instruct` | | mistral | `mistral-medium-3-5` | `openrouter` | `mistralai/mistral-medium-3-5` |
| mistral | `mistral-small-4` | `mistralai/mistral-small-2603` | | mistral | `mistral-small-3` | `openrouter` | `mistralai/mistral-small-3.2-24b-instruct` |
| nvidia | `nemotron-3-ultra` | `nvidia/nemotron-3-ultra-550b-a55b` | | mistral | `mistral-small-4` | `openrouter` | `mistralai/mistral-small-2603` |
| openai | `gpt-5-mini` | `openai/gpt-5.4-mini` | | nvidia | `nemotron-3-ultra` | `openrouter` | `nvidia/nemotron-3-ultra-550b-a55b` |
| openai | `gpt-5-nano` | `openai/gpt-5.4-nano` | | openai | `gpt-5-mini` | `openrouter` | `openai/gpt-5.4-mini` |
| openai | `gpt-5-nano` | `openrouter` | `openai/gpt-5.4-nano` |
## Schemas ## Schemas
@@ -285,16 +371,25 @@ schema produces a failed validation result.
Credential values belong at the request or environment boundary, never in Credential values belong at the request or environment boundary, never in
prompt, profile, schema, or example files: prompt, profile, schema, or example files:
- a file profile names an environment variable with `api_key_env`; - a backend or file profile can name an optional environment lookup source
- an in-memory profile may set `APIKeyRequired`; with `APIKeyEnv` or `api_key_env`;
- a request can provide a direct `APIKey` or override `APIKeyEnv`; and - an in-memory profile may set `APIKeyRequired` as an explicit local
requirement;
- a request can provide a direct `APIKey` or override the optional `APIKeyEnv`
source; and
- a direct request key takes precedence over environment lookup. - a direct request key takes precedence over environment lookup.
After a direct request key, the credential-source precedence is request After a direct request key, the credential-source precedence is request
`APIKeyEnv`, profile `api_key_env`, then the backend default. An in-memory `APIKeyEnv`, profile `api_key_env`, then the backend default. An in-memory
profile with `APIKeyRequired` clears an inherited backend environment name and profile with `APIKeyRequired` clears an inherited backend environment name and
requires a direct key unless the request explicitly supplies `APIKeyEnv`. requires a direct key unless the request explicitly supplies `APIKeyEnv`.
Promptkit validates required credential availability during preparation. Named environment sources are optional: when the selected source is absent,
empty, or whitespace-only, the built-in client omits the `Authorization`
header and handles the provider response normally. `APIKeyRequired` is the
only explicit local availability requirement. Promptkit validates required
credential availability during preparation and rechecks it when a prepared
execution runs. Injected clients receive resolved source metadata but define
their own credential-resolution behavior.
Direct keys are excluded from JSON results and redacted by public string Direct keys are excluded from JSON results and redacted by public string
formatters. Environment-variable names may appear in prepared metadata, but formatters. Environment-variable names may appear in prepared metadata, but
their values do not. their values do not.

View File

@@ -14,10 +14,16 @@ that produce these outbound settings.
Generation sends an HTTP `POST` with `Content-Type: application/json`. Generation sends an HTTP `POST` with `Content-Type: application/json`.
Before the client is called, the engine resolves framework, backend, profile, Before the client is called, the engine resolves framework, backend, profile,
and request values into one execution target. A non-empty endpoint from that and request values into one execution target. Endpoint configuration is trimmed
target overrides the client's configured base URL. After trailing slashes are and must be an absolute HTTP or HTTPS URL with a host and without user
removed, `/chat/completions` is appended. Generation fails before sending when information, a query, or a fragment. A non-empty endpoint from the target
neither source supplies an endpoint. overrides the client's configured base URL. The final selected endpoint is
validated again before transport.
The completion URL is composed through parsed URL path operations. Nested base
paths are retained, repeated trailing slashes are normalized, and the result
has exactly one appended `/chat/completions` suffix. Generation fails before
sending when neither source supplies a valid endpoint.
The target's backend ID is routing metadata for prepared values, results, and The target's backend ID is routing metadata for prepared values, results, and
injected clients. The built-in client does not derive the URL from that ID and injected clients. The built-in client does not derive the URL from that ID and
@@ -25,11 +31,13 @@ does not serialize it in the provider request.
## Authentication ## Authentication
A non-empty API key supplied directly on the execution target takes A usable API key supplied directly on the execution target takes precedence.
precedence. Otherwise, when an API-key environment-variable name is supplied, Otherwise, when an API-key environment-variable name is supplied, the client
the client reads that variable and requires a non-empty value. The selected reads and trims that variable. A bearer header is sent only when the resolved
key is sent as `Authorization: Bearer <key>`. No authorization header is sent direct or environment credential is non-empty. When neither source is usable,
when neither mechanism is configured. the client omits `Authorization` and handles the provider response normally.
An explicitly required target with no usable source is rejected before
transport.
The target contains the already resolved environment-variable name: an The target contains the already resolved environment-variable name: an
explicit request override takes precedence over profile metadata, which takes explicit request override takes precedence over profile metadata, which takes
@@ -47,18 +55,26 @@ Each ordinary message contains its `role` and string `content`. A
cache-controlled message instead uses a text content block containing `type`, cache-controlled message instead uses a text content block containing `type`,
`text`, and `cache_control`; an empty cache-control TTL is omitted. `text`, and `cache_control`; an empty cache-control TTL is omitted.
Promptkit sends only `developer`, `system`, `user`, and `assistant` roles and
does so without provider-specific translation. Tool and deprecated function
payloads are outside this text-message contract. A backend or model that
rejects an otherwise supported role or context returns its ordinary provider
error, which follows the normal generation-error path.
The effective direct or prompt-rendered session ID is trimmed, limited to 256 The effective direct or prompt-rendered session ID is trimmed, limited to 256
Unicode code points, and sent when nonempty as top-level `session_id`. It is Unicode code points, and sent when nonempty as top-level `session_id`. It is
never also sent as a session header. never also sent as a session header.
The client conditionally includes: The client conditionally includes:
- `temperature`, `max_tokens`, and `top_p` when non-zero or explicitly - `temperature`, `max_tokens`, and `top_p` only when selected by a profile or
present; runtime override, including an explicit runtime zero; they are absent when
unspecified;
- non-empty `service_tier` and effective `reasoning_effort`; an explicitly - non-empty `service_tier` and effective `reasoning_effort`; an explicitly
disabled reasoning setting is empty and therefore omitted; and disabled reasoning setting is empty and therefore omitted; and
- `response_format` for JSON Schema structured output, including its name, - `response_format` for JSON Schema structured output, including its name,
strict flag, and schema document. strict flag, and schema document. Plain JSON validation does not add an
object-only response constraint.
The engine resolves backend, profile, and request extra-parameter maps by The engine resolves backend, profile, and request extra-parameter maps by
whole-map replacement rather than key merging. The resulting effective map is whole-map replacement rather than key merging. The resulting effective map is
@@ -81,13 +97,47 @@ request fields.
## Response Handling ## Response Handling
Any 2xx response is decoded as an OpenAI-compatible chat response. The client Any 2xx response body is limited to 16 MiB (16,777,216 bytes). A larger
returns the first choice's non-empty message content and maps prompt, declared `Content-Length` is rejected before the body is read, and streamed,
completion, total, cached, and cache-write token counts. chunked, or underreported bodies are read through the same bound with at most
one additional byte used to detect overflow. A body exactly at the limit is
allowed. The body is closed on every outcome and an oversized stream is not
drained.
Invalid JSON, absent choices, and empty first-choice content are malformed The bounded body must contain exactly one OpenAI-compatible JSON response
responses. For a non-2xx status, the error includes the status code but never object followed only by JSON whitespace and EOF. The client returns the first
the provider response body. choice's explicitly present string message content, including an empty or
whitespace-only string, and maps prompt, completion, total, cached, and
cache-write token counts. Invalid or truncated JSON, trailing non-whitespace
data, a second JSON value, absent choices, missing content, `null` content,
non-string content, and size overflow are malformed responses and return no
partial result.
For a non-2xx status, Promptkit recognizes one JSON document with a top-level
object-valued `error` member. Its optional `message` and `type` fields must be
strings, and `code` may be a string or JSON number. Valid supported fields are
handled independently, numeric codes retain their JSON number text, and
unknown fields are ignored. Missing, invalid, malformed, or multiply framed
envelopes contribute no provider detail.
Non-success bodies have a 65,536-byte limit. A larger declared
`Content-Length` is not read; otherwise the client reads at most one additional
byte to detect streamed or underreported overflow. Empty, unreadable,
oversized, malformed, and unrecognized bodies retain only the received status.
The body is always closed and no oversized stream is drained beyond that probe.
Extracted strings are made valid UTF-8, trimmed, and converted to one line by
collapsing Unicode whitespace, control, and format-character runs. Blank
values are omitted. Codes and types longer than 256 Unicode code points are
omitted; messages longer than 4,096 code points are truncated at a code-point
boundary with an ellipsis inside the limit. Promptkit never exposes raw bodies,
headers, endpoints, credentials, request data, schemas, generated content, or
unsupported provider metadata through this handling.
An outbound `http.Client.Do` failure retains both Promptkit's request-failure
identity and the exact transport error for `errors.Is` and `errors.As` checks.
The rendered error does not include the selected endpoint, request headers,
request content, credentials, or provider body.
## Timeout And Cancellation ## Timeout And Cancellation
@@ -102,5 +152,7 @@ Timeouts are layered:
timeout when the supplied value is not positive. timeout when the supplied value is not positive.
The earliest applicable caller, generation, or transport deadline controls the The earliest applicable caller, generation, or transport deadline controls the
request. Constructing the internal client does not mutate a supplied request. Caller cancellation retains `context.Canceled`; caller, generation,
`http.Client`. and whole-request timeout failures retain `context.DeadlineExceeded`, together
with the request-failure identity. Constructing the internal client does not
mutate a supplied `http.Client`.

View File

@@ -29,27 +29,37 @@ One pool owns immutable active and total limits plus mutex-protected admission
count, active count, and ordered waiter list. Pool state exists only for the count, active count, and ordered waiter list. Pool state exists only for the
lifetime of its engine. lifetime of its engine.
## Bounded Run Admission ## Bounded Execution Admission
The runner asks the manager to admit a run after resolving the prompt, profile, For ordinary `Run`, the runner asks the manager to admit after resolving the
selected backend, effective execution target, credentials, and output contract, prompt, profile, selected backend, effective execution target, credentials, and
but before schema loading, artifact loading, or rendering. Admission is output contract, but before schema loading, artifact loading, or rendering.
immediate: a limited pool either reserves a slot or returns the internal `PrepareExecution` performs no admission. `RunPrepared` claims its handle,
`ErrCapacityExceeded` identity. The root facade maps that identity to the rechecks credential availability, and then asks the manager to admit the
public error without treating it as an invalid request or generation failure. frozen backend before generation.
Admission is immediate: a limited pool either reserves a slot or returns only
the internal `ErrCapacityExceeded` identity. The runner attaches the selected
backend identity at its use-case boundary, and the root facade translates that
typed value without treating it as an invalid request or generation failure.
The total admitted bound is the active-generation limit plus its configured The total admitted bound is the active-generation limit plus its configured
waiting capacity. The returned release function is idempotent. The runner waiting capacity. The returned release function is idempotent. The runner
defers it as soon as admission succeeds and holds the lease across remaining defers it as soon as admission succeeds. An ordinary run holds the lease across
preparation, initial generation, validation, every repair attempt, and all remaining preparation, initial generation, validation, every repair attempt,
failure or cancellation exits. A repair is part of its original admission and and all failure or cancellation exits. Prepared execution holds the normal
does not reserve another bounded slot. lease across generation, validation, every internal repair attempt, and all
execution exits. A repair is part of its original admission and does not
reserve another bounded slot.
## FIFO Generation Permits ## FIFO Generation Permits
`NewClient` wraps the engine's selected internal model client after public `NewClient` wraps the engine's selected internal model client after public
client adaptation or built-in client construction. Initial generation and the client adaptation or built-in client construction. Initial generation and the
default repairer receive the same wrapper. default repairer receive the same wrapper. Their requests retain the same
effective backend, credential, numeric-presence metadata, and structured-output
settings, so scheduling does not change provider omission semantics between
calls.
For each `Generate` call, the wrapper selects a pool from the request's For each `Generate` call, the wrapper selects a pool from the request's
effective backend ID. An unlimited call passes directly to the next client. A effective backend ID. An unlimited call passes directly to the next client. A
@@ -64,7 +74,9 @@ other backend IDs.
The wrapper passes generation requests, responses, and collaborator errors The wrapper passes generation requests, responses, and collaborator errors
through unchanged. It owns scheduling only; the concrete model client remains through unchanged. It owns scheduling only; the concrete model client remains
responsible for provider transport behavior. responsible for provider transport behavior. The runner, rather than the
capacity layer, sums all five usage fields from the initial response and every
completed repair response into the successful run result.
## Cancellation And Release ## Cancellation And Release
@@ -90,11 +102,17 @@ unlimited admission. The
FIFO transfer, canceled-waiter removal, grant/cancel races, independent pools, FIFO transfer, canceled-waiter removal, grant/cancel races, independent pools,
unlimited calls, passthrough behavior, and panic release. unlimited calls, passthrough behavior, and panic release.
The [runner tests](../../internal/usecase/runner_test.go) own early admission, The [runner tests](../../internal/usecase/runner_test.go) own ordinary early
lease lifetime, failure release, and shared initial/repair scheduling. The admission, lease lifetime, failure release, and shared initial/repair
scheduling. The
[prepared-execution use-case tests](../../internal/usecase/prepared_execution_test.go)
own deferred admission, credential ordering, and prepared-execution lease
release. The
[external package capacity tests](../../capacity_contract_test.go) own the [external package capacity tests](../../capacity_contract_test.go) own the
assembled public-engine behavior for configured limits, capacity errors, assembled public-engine behavior for configured limits, capacity errors,
endpoint identity, engine independence, and injected clients. The endpoint identity, engine independence, and injected clients. The
[prepared-execution contract tests](../../prepared_execution_contract_test.go)
own the public prepared-capacity boundary. The
[root error-boundary tests](../../errors_internal_test.go) own preservation of [root error-boundary tests](../../errors_internal_test.go) own preservation of
the public generation category and context identity when generation is the public generation category and context identity when generation is
canceled. canceled.

View File

@@ -21,20 +21,26 @@ uses internal domain values for rendered prompts, execution targets,
structured output, responses, and token usage. structured output, responses, and token usage.
The runner supplies a fully resolved target after applying backend, profile, The runner supplies a fully resolved target after applying backend, profile,
and request precedence. The client uses its endpoint, credential metadata, and request precedence, plus canonical provider-bound text messages. The
generation fields, and extra parameters. `BackendID` remains routing metadata client uses its endpoint, credential metadata, generation fields, and extra
parameters. `BackendID` remains routing metadata
for the generation boundary and is not mapped into the provider payload. for the generation boundary and is not mapped into the provider payload.
Construction validates the configured base URL and clones any supplied Construction trims and validates a nonempty configured base URL and clones any
`http.Client` so Promptkit can apply its timeout default without mutating the supplied `http.Client` so Promptkit can apply its timeout default without
caller's client. Generation then: mutating the caller's client. An empty configured base remains valid because a
resolved request target may supply the endpoint. Generation then:
1. validates request-level timeout and endpoint requirements; 1. validates shared execution-setting invariants and the final selected base
endpoint;
2. maps the internal request into the OpenAI-compatible chat payload; 2. maps the internal request into the OpenAI-compatible chat payload;
3. validates and merges extra parameters; 3. validates and merges extra parameters;
4. resolves authentication; 4. composes `/chat/completions` through parsed URL path operations;
5. performs the outbound request under the applicable deadlines; and 5. resolves authentication;
6. decodes the first response choice and token usage. 6. performs the outbound request under the applicable deadlines; and
7. decodes one strictly framed, size-bounded successful response object and
maps its first choice and token usage, or decodes bounded structured
non-success detail.
`internal/llm` owns the set of reserved OpenAI-compatible request fields used `internal/llm` owns the set of reserved OpenAI-compatible request fields used
when validating extra parameters. Backend registration consumes the same rule when validating extra parameters. Backend registration consumes the same rule
@@ -43,16 +49,69 @@ without making the model client depend on registry configuration.
The implementation has no retry loop, tool-call support, provider catalog, The implementation has no retry loop, tool-call support, provider catalog,
inbound HTTP behavior, or durable session store. inbound HTTP behavior, or durable session store.
## Prepared Generation
For [`RunPrepared`](../../engine.go), the runner supplies the model client with
the target, rendered messages, and structured-output constraint retained by
executable preparation. Execution does not reopen or rerender consumer
sources.
Before backend admission, the runner rechecks a frozen credential
environment-variable name only when the target explicitly requires a
credential. The handle does not retain the environment value; the model client
resolves the value visible when generation begins. For optional sources with no
usable value, the built-in client omits `Authorization` and continues to the
provider. A direct request key remains in private execution state only until
the claimed execution finishes or an unclaimed handle is discarded. Exact
public ownership and redaction semantics belong to the
[`PreparedExecution` GoDoc](../../prepared_execution.go).
## Failure Categories ## Failure Categories
The package preserves distinct error identities for invalid client The package preserves distinct error identities for invalid client
configuration, invalid generation requests, request execution failures, configuration, invalid generation requests, request execution failures,
non-success provider statuses, and malformed successful responses. Provider non-success provider statuses, and malformed successful responses. Provider
response bodies are not included in non-success errors. response bodies are never exposed in raw form through non-success errors.
Caller cancellation and deadline failures during the outbound request are Invalid nonempty configured endpoints are configuration failures. A missing or
reported as request execution failures. The runner classifies these identities invalid final selected endpoint is an invalid generation request and is
without depending on HTTP status mapping. rejected before transport.
Authentication resolves a trimmed direct key before a trimmed configured
environment value. Optional missing, empty, or whitespace-only sources do not
block transport and produce no `Authorization` header. An explicitly required
target with no usable source is rejected before transport with the existing
invalid-request diagnostics.
Successful response bodies have a fixed 16 MiB limit enforced by declared
length and by reading at most one byte beyond the boundary. The decoder accepts
exactly one JSON object plus trailing whitespace and EOF. Size overflow,
truncation, malformed JSON, trailing data, and a second value are malformed
responses with no partial result or provider content in the error. Every body
is closed, and an unbounded oversized stream is not drained.
After framing succeeds, the first choice must contain an explicitly present
string `message.content`. The string is returned exactly, including empty or
whitespace-only content. Missing choices, missing or `null` content, and
non-string content are malformed responses. Output validation and correction
eligibility remain outside this package.
For a non-success response, `ProviderHTTPError` retains the HTTP status and
only normalized detail from the bounded recognized envelope. It retains
`ErrUnexpectedStatus` through unwrapping. The client owns response closure;
its bounded reader and parser never close or drain a body themselves. The root
facade converts this concrete internal error into the public
[`GenerationError`](../../generation_error.go), while arbitrary injected-client
errors continue through the ordinary generation-error mapping unchanged.
An `http.Client.Do` failure is represented by a redacting multi-cause error:
the package request-failure sentinel and the exact returned transport error are
both available through `errors.Is` and `errors.As`, while the rendered text
does not expose the endpoint, headers, request content, credential, transport
detail, or provider body. Caller cancellation retains `context.Canceled`;
caller deadlines, generation deadlines, and whole-request client timeouts
retain `context.DeadlineExceeded`. The runner adds its generation category
without discarding those identities or depending on HTTP status mapping.
## Test Ownership ## Test Ownership
@@ -60,7 +119,15 @@ The
[OpenAI-compatible client tests](../../internal/llm/openai_compatible_client_test.go) [OpenAI-compatible client tests](../../internal/llm/openai_compatible_client_test.go)
own configuration, client cloning, deterministic deadline precedence, own configuration, client cloning, deterministic deadline precedence,
authentication, request and response mapping, malformed data, error identity, authentication, request and response mapping, malformed data, error identity,
cancellation, and response-body suppression. The root transport contract test cancellation, endpoint selection and composition, pre-transport rejection, and
also verifies that resolved backend settings reach this client without bounded single-document successful-response framing, closure, and
serializing backend identity. All use local test servers or test transports; response-body suppression. The focused
the default suite makes no live or paid provider requests. [provider HTTP error tests](../../internal/llm/provider_http_error_test.go)
own envelope parsing, normalization, and bounded-reader cases; their
[transport tests](../../internal/llm/provider_http_error_transport_test.go)
own non-success response closure and integration. Root transport contract tests
own public `GenerationError` conversion, while also verifying that resolved
backend settings reach this client without serializing backend identity and
that ordinary-run cancellation retains its public generation and context
identities. All use local test servers or controlled test transports; the
default suite makes no live or paid provider requests.

View File

@@ -11,23 +11,23 @@ contributor workflow and validation.
| Component | Implemented responsibility | References | | Component | Implemented responsibility | References |
| --- | --- | --- | | --- | --- | --- |
| Root `promptkit` package | Provides the supported engine facade, source, backend-registration, and injection options, public request and result values, profile construction, extension interfaces, value conversion, redacted formatting, public error mapping, and engine-local assembly. | [Package GoDoc](../../doc.go), [backend API](../../backends.go), [engine assembly](../../engine.go) | | Root `promptkit` package | Provides the supported engine facade, source, backend-registration, and injection options, public request, result, prompt-inspection, and profile-inspection values, opaque prepared-execution handles, profile construction, extension interfaces, value conversion, redacted formatting, typed capacity and generation error mapping, engine-local profile-source assembly including application fallbacks, and bounded output-repair assembly. | [Package GoDoc](../../doc.go), [prepared execution](../../prepared_execution.go), [backend API](../../backends.go), [engine assembly](../../engine.go) |
| `examples/go-library/prepare` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, and `Prepare`. It is not a public library package. | [Example program](../../examples/go-library/prepare/main.go) | | `examples/go-library/prepare` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, and `Prepare`. It is not a public library package. | [Example program](../../examples/go-library/prepare/main.go) |
| `examples/go-library/run` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, an injected deterministic model client, and `Run`. It is not a public library package. | [Example program](../../examples/go-library/run/main.go) | | `examples/go-library/run` | Demonstrates an offline downstream consumer using a prompt file, in-memory profile, inline input, an injected deterministic model client, and `Run`. It is not a public library package. | [Example program](../../examples/go-library/run/main.go) |
| `internal/backend` | Constructs each engine's immutable registry from the built-in OpenRouter definition and consumer additions, validates and defensively copies definitions through the shared JSON-value package, and consumes the LLM-owned OpenAI-compatible reserved request-field rule. | [Backend registry](../../internal/backend/registry.go) | | `internal/backend` | Constructs each engine's immutable registry from maintained definitions and consumer additions, validates and defensively copies definitions through the shared JSON-value package, and consumes the LLM-owned OpenAI-compatible reserved request-field rule. | [Backend registry](../../internal/backend/registry.go) |
| `internal/capacity` | Owns engine-local bounded run admission and FIFO model-generation permits for limited backend IDs, including cancellation-safe waiter removal and client wrapping. | [Internal capacity management](capacity.md) | | `internal/catalog` | Strictly validates imported immutable maintained backend and profile catalog assets before engine assembly uses them. | [Catalog adapter](../../internal/catalog/catalog.go), [internal sources](sources.md#profiles-and-built-ins) |
| `internal/domain` | Defines internal framework values for requests, artifacts, prompt definitions, profiles, execution targets, rendering, generation, and validation. | [Domain declarations](../../internal/domain/domain.go) | | `internal/capacity` | Owns engine-local bounded execution admission and FIFO model-generation permits for limited backend IDs, including cancellation-safe waiter removal and client wrapping. | [Internal capacity management](capacity.md) |
| `internal/domain` | Defines internal framework values for requests, artifacts, prompt definitions, profiles, execution targets, rendering, generation, and validation, and owns source-neutral invariants for shared execution settings, OpenAI-compatible base endpoints, session identifiers, and output contracts. Source parsing, required fields, other source-specific normalization, defaulting, and boundary-specific error classification remain with their callers. | [Domain declarations](../../internal/domain/domain.go), [endpoint invariant](../../internal/domain/endpoint.go) |
| `internal/defaults` | Defines application-neutral framework constants and constructs the default execution target. It contains no CLI, server, or inbound HTTP limits. | [Framework defaults](../../internal/defaults/defaults.go) | | `internal/defaults` | Defines application-neutral framework constants and constructs the default execution target. It contains no CLI, server, or inbound HTTP limits. | [Framework defaults](../../internal/defaults/defaults.go) |
| `internal/filecatalog` | Provides deterministic YAML discovery and path helpers for operating-system filesystems and `fs.FS` sources. | [File catalog](../../internal/filecatalog/catalog.go) | | `internal/filecatalog` | Provides deterministic YAML discovery and path helpers for operating-system filesystems and `fs.FS` sources. | [File catalog](../../internal/filecatalog/catalog.go) |
| `internal/jsonvalue` | Validates and deeply copies JSON-compatible extra-parameter trees while preserving supported concrete value types. | [JSON values](../../internal/jsonvalue/jsonvalue.go) | | `internal/jsonvalue` | Validates and deeply copies bounded JSON-compatible extra-parameter and prepared-schema trees while preserving supported concrete value types and rejecting cycles or excessive depth and work. | [JSON values](../../internal/jsonvalue/jsonvalue.go) |
| `internal/promptdef` | Loads strictly decoded, validated prompt definitions from filesystem and `fs.FS` sources, including version selection and contained file-backed message content. | [Framework formats](../formats.md), [prompt-definition repository](../../internal/promptdef/filesystem_repository.go) | | `internal/promptdef` | Loads strictly decoded, validated prompt definitions from filesystem and `fs.FS` sources, including version selection and contained file-backed message content. | [Framework formats](../formats.md), [prompt-definition repository](../../internal/promptdef/filesystem_repository.go) |
| `internal/profile` | Loads strictly decoded, validated execution profiles, including backend selection, from filesystem and `fs.FS` sources and composes repositories with error-preserving fallback. | [Framework formats](../formats.md), [profile repositories](../../internal/profile/filesystem_repository.go) | | `internal/profile` | Loads strictly decoded, locally validated execution profiles from filesystem and `fs.FS` sources, overlays raw sources with error-preserving fallback, and resolves inherited profiles. | [Framework formats](../formats.md), [profile repositories](../../internal/profile/filesystem_repository.go), [internal sources](sources.md#profiles-and-built-ins) |
| `internal/profile/builtin` | Embeds the built-in profile catalog, whose entries select OpenRouter, and combines it with an optional primary repository. | [Built-in catalog](../formats.md#built-in-profile-catalog), [repository](../../internal/profile/builtin/repository.go) |
| `internal/prompt` | Renders prompt messages from Go templates with artifact, variable, session, and cache-control data. | [Go-template renderer](../../internal/prompt/go_renderer.go) | | `internal/prompt` | Renders prompt messages from Go templates with artifact, variable, session, and cache-control data. | [Go-template renderer](../../internal/prompt/go_renderer.go) |
| `internal/artifact` | Resolves ordinary inline and unrestricted caller-selected file references into copied artifacts with metadata and hashes. | [Internal sources and validation](sources.md) | | `internal/artifact` | Resolves ordinary inline and unrestricted caller-selected file references into copied artifacts with metadata and hashes. | [Internal sources and validation](sources.md) |
| `internal/validate` | Validates basic, JSON, and JSON Schema output using operating-system filesystem or `fs.FS` schema sources. | [Framework formats](../formats.md#schemas), [internal sources and validation](sources.md) | | `internal/validate` | Validates basic, JSON, and JSON Schema output using operating-system filesystem or `fs.FS` schema sources and creates operation-local validation plans with canonical contained schema resources. | [Framework formats](../formats.md#schemas), [internal sources and validation](sources.md) |
| `internal/llm` | Defines the internal generation boundary and implements outbound OpenAI-compatible chat requests from resolved execution targets, including response decoding, authentication, deadline handling, and ownership of the OpenAI-compatible reserved request-field policy. | [Internal model client](llm.md) | | `internal/llm` | Defines the internal generation boundary and implements outbound OpenAI-compatible chat requests from resolved execution targets, including bounded structured non-success response decoding, successful-response decoding, authentication, deadline handling, and ownership of the OpenAI-compatible reserved request-field policy. | [Internal model client](llm.md) |
| `internal/usecase` | Resolves backend, profile, and request settings and coordinates preparation and execution across internal sources, rendering, artifact loading, generation, validation, and optional repair. | [Internal runner](runner.md) | | `internal/usecase` | Resolves prompt definitions and hashes, profiles, backends, and targets for exact inspection and request settings for preparation, and coordinates ordinary execution and one-attempt prepared execution across internal sources, rendering, artifact loading, operation-local validation plans, generation, capacity, and bounded repair. | [Internal runner](runner.md), [prepared-execution implementation](../../internal/usecase/prepared_execution.go) |
The root package assembles these internal components without exposing their The root package assembles these internal components without exposing their
representations. Consumers depend only on the root facade. representations. Consumers depend only on the root facade.

View File

@@ -20,14 +20,45 @@ and override semantics consumed by the runner.
profiles, backend resolution, artifacts, rendering, model generation, and profiles, backend resolution, artifacts, rendering, model generation, and
validation. The root engine supplies one immutable registry containing the validation. The root engine supplies one immutable registry containing the
built-in backend and validated consumer additions, one engine-local run built-in backend and validated consumer additions, one engine-local run
admitter, and a model client wrapped by the same capacity manager. Schema admitter, and a model client wrapped by the same capacity manager. Validation
documents are loaded through the validator's optional schema-loader interface. plans and provider-facing schema metadata come from the validator's preparation
An output repairer can be injected internally, but the ordinary runner interface.
constructor does not enable one. The root engine supplies one default output repairer through the explicit
runner constructor, using the same capacity-wrapped client as initial
generation. The no-repair runner constructor remains available for focused
internal callers and tests.
Each invocation carries its state in request, prepared-run, and result values. Each invocation carries its state in request, prepared-run, and result values.
The runner has no durable run or session store. The runner has no durable run or session store.
## Shared Prompt Selection
The runner uses one prompt-selection and hashing boundary for ordinary
preparation and exact prompt inspection. Preparation retains its early
request-ID check before direct-session normalization; both operations then use
the configured prompt repository to select one definition, load referenced
message content, and calculate the same prompt hash.
Inspection stops after that structural lookup. It does not parse templates or
touch profile, artifact, schema, renderer, validator, admission, or model
collaborators. The root [`Engine.InspectPrompt`](../../engine.go) GoDoc owns
the public operation's exact contract.
## Shared Profile Selection
The runner uses one profile-selection and target-resolution boundary for
ordinary preparation and exact profile inspection. Preparation first selects a
request profile or a prompt default; inspection begins with its required
explicit profile ID. Both then apply the ordinary source precedence, resolve a
named backend, and construct the effective target from framework, backend, and
profile values.
Inspection stops after the resulting endpoint and model are structurally
validated. It does not check credential availability or perform prompt,
artifact, schema, rendering, admission, or model-client work. The root
[`Engine.InspectProfile`](../../engine.go) GoDoc owns the public operation's
exact contract.
## Shared Preparation Pipeline ## Shared Preparation Pipeline
`Prepare` and `Run` share one private preparation pipeline split at the point `Prepare` and `Run` share one private preparation pipeline split at the point
@@ -41,17 +72,20 @@ performs only the work needed to validate routing and admission:
5. resolve application-neutral defaults, backend defaults, profile values, 5. resolve application-neutral defaults, backend defaults, profile values,
and explicit request overrides in that order; and explicit request overrides in that order;
6. validate endpoint, model, numeric overrides, and credential requirements; 6. validate endpoint, model, numeric overrides, and credential requirements;
7. resolve the effective output contract without loading its schema; and 7. resolve and validate the effective output contract without loading its
schema; and
8. retain the definition, source identities, effective settings, output 8. retain the definition, source identities, effective settings, output
contract, and preparation start time in invocation-local state. contract, and preparation start time in invocation-local state.
The completion phase consumes that state without reloading the prompt, The completion phase consumes that state without reloading the prompt,
profile, or backend: profile, or backend:
1. load structured-output schema metadata when required; 1. create one operation-local validation plan and derive structured-output
schema metadata from it when required;
2. load and hash input artifacts; 2. load and hash input artifacts;
3. render messages and the prompt-defined session; 3. render messages and the prompt-defined session;
4. apply any direct session ID; 4. apply any direct session ID, then append already-normalized request messages
after the rendered definition messages;
5. hash the effective rendered prompt; and 5. hash the effective rendered prompt; and
6. construct the prepared value and preparation timing. 6. construct the prepared value and preparation timing.
@@ -59,7 +93,10 @@ profile, or backend:
`Run` performs backend admission between the phases. This structure preserves `Run` performs backend admission between the phases. This structure preserves
one execution-precedence and error-ordering implementation while allowing a one execution-precedence and error-ordering implementation while allowing a
full backend pool to reject work before expensive schema, artifact, and full backend pool to reject work before expensive schema, artifact, and
rendering operations. rendering operations. `Prepare` discards the plan after returning its public
metadata. `Run` retains the plan through initial and repaired-output validation
and discards it when the operation ends. Prepared execution stores the same
kind of plan only in its private payload.
Pointer-based numeric overrides preserve an explicit zero. Invalid negative or Pointer-based numeric overrides preserve an explicit zero. Invalid negative or
out-of-range values fail as invalid requests. Endpoint overrides do not change out-of-range values fail as invalid requests. Endpoint overrides do not change
@@ -77,7 +114,10 @@ the prompt session template, and is applied after ordinary message rendering.
A blank direct value retains prompt-template behavior. The runner clears the A blank direct value retains prompt-template behavior. The runner clears the
template only on a value copy of the definition, so the definition hash always template only on a value copy of the definition, so the definition hash always
describes the original source while the rendered-prompt hash includes the describes the original source while the rendered-prompt hash includes the
effective direct or rendered session. effective direct or rendered session and the complete effective message
sequence. The rendered-prompt hash uses a versioned, length-framed SHA-256
encoding of that session, every role and content value, and cache-control
presence and values; its hexadecimal value is opaque.
The registry is read-only after engine construction. Concurrent `Prepare` and The registry is read-only after engine construction. Concurrent `Prepare` and
`Run` calls resolve independent defensive backend values and keep all `Run` calls resolve independent defensive backend values and keep all
@@ -90,8 +130,17 @@ its `RunAdmitter` to reserve capacity for the effective backend ID. A nil
admitter is an internal unlimited fallback. After successful admission, `Run` admitter is an internal unlimited fallback. After successful admission, `Run`
immediately defers the returned release function, performs the completion immediately defers the returned release function, performs the completion
phase, makes one initial generation call, builds the named output artifact, phase, makes one initial generation call, builds the named output artifact,
and validates that artifact. Invalid generated content remains a validation and validates that artifact with the plan compiled during completion. Invalid
result; an inability to perform validation is an operational error. generated content remains a validation result; an inability to generate or
perform validation is an operational error.
Validation preparation and execution honor cancellation at every
Promptkit-controlled boundary and do not publish a partial plan or result.
Schema reads are bounded and context-checked between chunks; JSON decoding,
schema compilation, and schema execution are checked immediately before and
after their synchronous calls. Promptkit does not move arbitrary filesystem or
JSON Schema work to background goroutines, so an already-blocked dependency
method must return before cancellation can take precedence over its outcome.
The admission lease covers completion-phase preparation, initial generation, The admission lease covers completion-phase preparation, initial generation,
validation, every repair, and every exit. It bounds accepted work without validation, every repair, and every exit. It bounds accepted work without
@@ -99,22 +148,38 @@ serializing preparation or validation behind the active-generation limit.
The wrapped model client separately acquires a FIFO active permit only around The wrapped model client separately acquires a FIFO active permit only around
each actual generation call. each actual generation call.
When an internal repairer is present, a JSON or JSON Schema content failure can After a failed `basic`, JSON, or JSON Schema validation with a positive frozen
trigger bounded repair attempts. Repair receives the effective execution budget, the installed repairer can make a bounded corrective call. Each request
target and session ID, validation errors, prior output, and structured-output starts with a fresh copy of the complete effective message sequence (the
specification. The default repairer uses the same wrapped client as initial configured prefix followed by the request suffix), includes only the latest
nonempty candidate as an assistant message, and appends one corrective user
message. Empty candidates omit that assistant message. The
correction carries validation diagnostics as JSON data bounded to 64 KiB; the
full diagnostics remain in the validation result.
Repair receives the effective execution target, explicit numeric-presence bits,
credential, backend identity, session ID, and structured-output specification.
The same request constructor supplies those common fields to initial and repair
generation. The default repairer uses the same wrapped client as initial
generation, so each repair reacquires the selected backend's active permit generation, so each repair reacquires the selected backend's active permit
while remaining inside its original admission lease. Repair never performs a while remaining inside its original admission lease. Repair never performs a
second bounded admission. This capability remains internal and is not a public second bounded admission, and repaired outputs use the operation's existing
option. validation plan. The runner stops at the first valid candidate, sums completed
generation usage, reports calls actually started, and returns the final failed
validation result on exhaustion. A repair generation failure follows the
ordinary generation-error category rather than becoming a validation error.
A successful result includes the output artifact and raw output, validation A successful result includes the output artifact and raw output, validation
state, effective session ID, prompt and rendered-prompt hashes, selected state, effective session ID, prompt and rendered-prompt hashes, selected
profile and backend, effective settings, input hashes, token usage, a generated profile and backend, effective settings, input hashes, token usage, a generated
run identifier, and UTC timing. The same effective session reaches initial run identifier, and UTC timing. The same effective session reaches initial
generation and any repair attempt through the rendered prompt. The same generation and any repair attempt through the rendered prompt. The same
effective target, including backend identity, reaches generation and any effective target and presence metadata, including backend identity and direct
repair attempt. credential during execution, reaches generation and every repair attempt.
Result usage is the field-wise sum of all five usage values from the initial
response and every completed repair response. Final raw output, artifact, and
validation state still come from the last candidate. A repair error returns no
partial run result or partial usage.
## Failure Categories ## Failure Categories
@@ -124,18 +189,21 @@ validation failures. Wrapping preserves the package identities mapped by the
public facade and retains collaborator identities where they are part of the public facade and retains collaborator identities where they are part of the
internal contract. internal contract.
Admission capacity exhaustion retains the internal capacity identity and adds Admission capacity exhaustion retains the internal capacity identity. At the
the selected backend ID as context. It is not recategorized as an invalid use-case boundary, the runner attaches the selected backend ID in an internal
request or generation failure, and no partial result is returned. A context typed error, and the root facade copies that value into the public
already done at admission retains its context identity directly. Cancellation [`CapacityError`](../../capacity_error.go) without parsing diagnostic text. It
while waiting for an active generation permit prevents client invocation when is not recategorized as an invalid request or generation failure, and no
it wins the grant race; the model-client boundary then preserves the context partial result is returned. A context already done at admission retains its
error through the generation-failure category. Deferred release restores the context identity directly. Cancellation while waiting for an active generation
admission lease on preparation, generation, validation, repair, and permit prevents client invocation when it wins the grant race; the model-client
cancellation failures. boundary then preserves the context error through the generation-failure
category. Deferred release restores the admission lease on preparation,
generation, validation, repair, and cancellation failures.
Other context cancellation propagates through the invoked collaborator and is Other context cancellation propagates through the invoked collaborator and is
classified by the owning operation. classified by the owning operation. In particular, cancellation observed by
validation retains the context identity through the validation error category.
An overlong direct session is an invalid request before source loading, while An overlong direct session is an invalid request before source loading, while
an invalid or overlong prompt session template remains a prompt-render failure. an invalid or overlong prompt session template remains a prompt-render failure.
An unknown selected backend, or a selected backend with no configured resolver, An unknown selected backend, or a selected backend with no configured resolver,
@@ -147,8 +215,9 @@ The [runner tests](../../internal/usecase/runner_test.go) own preparation order,
selection and override precedence, the two-phase boundary, early admission, selection and override precedence, the two-phase boundary, early admission,
lease lifetime and release, direct-session resolution, schema-before-generation lease lifetime and release, direct-session resolution, schema-before-generation
behavior, hashing, generation and validation outcomes, backend propagation, behavior, hashing, generation and validation outcomes, backend propagation,
bounded repair, shared initial/repair capacity, credentials and redaction, bounded repair progression, initial/repair request parity, cumulative usage,
error categories, artifact metadata, usage, and timing. The shared initial/repair capacity, credentials and redaction, error categories,
artifact metadata, and timing. The
[capacity subsystem document](capacity.md) identifies the focused pool, [capacity subsystem document](capacity.md) identifies the focused pool,
waiter, and wrapped-client tests. waiter, and wrapped-client tests.

View File

@@ -12,9 +12,32 @@ validation modes, built-in catalog, and source precedence.
## Prompt Definitions ## Prompt Definitions
`internal/promptdef` discovers YAML deterministically, decodes and validates `internal/promptdef` uses one source-neutral flow for prompt selection and
definitions, selects an ID and optional version, and resolves file-backed normalization. That flow scans normalized YAML ID and version metadata,
message content within the selected operating-system or `fs.FS` source. requires one strictly decoded document per file, classifies errors for the
selected definition, detects duplicates, and normalizes the exact match.
Small operating-system and `fs.FS` adapters own discovery, byte reads, display
paths, content opening, and root containment. Each lookup remains a
point-in-time scan: definitions and catalogs are not cached, and file-backed
message content is opened only for the exact selected candidate.
Message roles are normalized through the shared domain owner by trimming
Unicode whitespace and lowercasing. Only `developer`, `system`, `user`, and
`assistant` are published; invalid roles remain selected prompt-definition
failures rather than becoming request errors. Cache-control metadata uses the
same shared domain normalization and defensive-copy rule.
Operating-system sources enforce containment against canonical roots and
targets so symlinks cannot escape. Injected `fs.FS` sources enforce containment
in their clean relative path namespace. A single-file source uses the selected
prompt file's containing directory as its root. Every content path must be
relative and is opened from its exact parsed text after a separate blank check;
contained parent components and whitespace-bearing names remain valid.
Exact prompt inspection performs one point-in-time lookup through that same
repository and validates referenced message content before returning declared
metadata. It does not parse templates or read profile, input, or schema
sources, and it does not retain the definition for a later execution.
Its package tests own prompt selection, strict decoding, definition validation, Its package tests own prompt selection, strict decoding, definition validation,
duplicate detection, and source containment: duplicate detection, and source containment:
@@ -22,30 +45,78 @@ duplicate detection, and source containment:
## Profiles And Built-Ins ## Profiles And Built-Ins
`internal/profile` loads and validates execution profiles from an `internal/profile` loads, locally validates, overlays, and resolves execution
operating-system filesystem or an `fs.FS`. It supports a primary repository profiles from an operating-system filesystem or an `fs.FS`. A file contains
with fallback only when the primary reports that a profile is absent. Strict exactly one YAML document and its trimmed YAML `id` is its only selection
YAML decoding recognizes the optional `backend` field, trims its value, and identity; filenames do not confer authority. Each point lookup reads discovered
requires a model plus at least one non-blank backend or endpoint. Loading does files once for their metadata and reuses the selected file's bytes for strict
not check registry membership because the available registry belongs to the decoding; unrelated profiles are not fully decoded. Strict selected decoding
assembled engine; the runner checks membership during preparation. recognizes `base_profile` and the optional `backend` field, trims their values,
and permits inherited target fields only when a base is named. File-backed
`extra_params` values are validated and defensively copied through the shared
bounded JSON-value owner before a profile is published. OpenAI-compatible
reserved-field policy remains with the model-client and backend-registry owners.
`internal/profile/builtin` embeds the maintained built-in profile catalog and `LoadFSRepository` is the eager immutable loading boundary for internal
can place a caller-selected repository ahead of that catalog. Every embedded catalog consumers. It discovers and strictly validates every raw profile once,
profile selects `openrouter` and inherits its endpoint and credential preserves safe source metadata including explicitly present YAML fields, and
environment-variable name from the built-in backend registry rather than publishes independently copied values from memory. It does not resolve profile
repeating those values. Profile behavior is owned by the inheritance. Configured consumer sources continue to use the lazy point lookup
repositories described above.
`internal/catalog` validates the imported immutable OpenRouter and
Rakestrawhome asset modules as one private adapter boundary. It enforces their
manifest, layout, profile ownership, inheritance, and secret-safety rules
before returning raw catalog sources. Root assembly uses those validated
catalogs as the maintained lowest-precedence profile source.
The overlay repository consults the next repository only when the
higher-precedence repository reports that a profile is absent. A reliably
selected malformed profile stops fallback, while an unrelated malformed file
does not become authoritative through its filename. Loading does not check
backend registry membership because the available registry belongs to the
assembled engine; the runner checks membership during preparation and exact
profile inspection.
The root engine assembles one raw composite catalog in precedence order:
in-memory profiles, one ordinary configured source, an application fallback
source, then the maintained external catalog. An explicit file or `fs.FS` profile
source replaces `Config.ProfileDir` within the ordinary configured-source
category. One outer resolving repository wraps that complete raw catalog, so
each base lookup observes the same precedence and shadowing rules.
The resolving repository traverses every selected chain afresh, retains no
cache, detects cycles, limits a chain to 32 profiles, merges root-to-leaf into a
new caller-owned value, and validates the final target before publishing it. It
does not check backend registry membership. Exact `base_profile` syntax, merge
rules, and consumer-visible failure behavior belong to the [framework format
reference](../formats.md#profile-inheritance).
Exact profile inspection performs one point-in-time resolved lookup through
those profile sources and checks the final target without reading prompt, input,
or schema sources. It does not retain that lookup for a later execution.
Prepared execution instead freezes the fully resolved target; a later ordinary
operation performs a fresh traversal.
The maintained external catalog provides every built-in profile and its
matching backend definition. Profile loading and overlay behavior are owned by the
[profile repository tests](../../internal/profile/repository_test.go), while [profile repository tests](../../internal/profile/repository_test.go), while
catalog completeness, the backend-selection invariant, duplicate IDs, and catalog completeness, backend selection, and duplicate IDs are owned by the
overlay behavior are owned by the [catalog adapter tests](../../internal/catalog/catalog_test.go).
[built-in repository tests](../../internal/profile/builtin/repository_test.go).
## Ordinary Artifacts ## Ordinary Artifacts
`internal/artifact` resolves inline references and unrestricted, `internal/artifact` accepts explicitly typed inline references even when their
caller-selected file paths. It copies content into an artifact, records body is empty. It also resolves unrestricted, caller-selected paths only when
metadata and a content hash, applies a content-type fallback, and honors they identify regular operating-system files, checking that condition before
context cancellation. and after opening the file. It copies content into an artifact, records
metadata and an opaque content-equality value, and applies a content-type
fallback.
Regular files are read synchronously in bounded chunks. Cancellation is
checked before opening, before and after every read, and before publishing the
artifact, so a canceled read never publishes partial content. The ordinary
reader does not detach file reads into background goroutines.
This ordinary reader does not implement an inbound HTTP security boundary. In This ordinary reader does not implement an inbound HTTP security boundary. In
particular, it does not constrain files to an application root or impose an particular, it does not constrain files to an application root or impose an
@@ -57,7 +128,17 @@ implemented reader behavior and failures.
## Rendering ## Rendering
`internal/prompt` renders definition messages as Go templates using named `internal/prompt` renders definition messages as Go templates using named
artifacts and variables. It carries message roles, session IDs, and cache artifacts and variables. Within one render, each referenced artifact body is
converted to text lazily and cached by input name for reuse across the session
and every message; the cache is not shared across renders. Conversion uses
bounded chunks and preserves the artifact bytes exactly.
Session and message parsing and execution remain synchronous. The renderer
checks cancellation before and after each parse and execution boundary,
between artifact conversion chunks, around each message, and before publishing
the complete prompt. It cannot interrupt template work already in progress and
never publishes a partial prompt after observing cancellation. It validates and
canonicalizes message roles before carrying roles, session IDs, and cache
control into the rendered prompt. The control into the rendered prompt. The
[renderer tests](../../internal/prompt/renderer_test.go) own rendering behavior. [renderer tests](../../internal/prompt/renderer_test.go) own rendering behavior.
@@ -68,6 +149,35 @@ filesystem or an `fs.FS`. Invalid generated content is returned as a validation
result; inability to load, register, or compile a schema is an operational result; inability to load, register, or compile a schema is an operational
error. error.
Every preparation operation creates one operation-local validation plan. None,
basic, and JSON modes retain the effective output contract without source
access. JSON Schema mode loads the root document once, resolves and compiles
each transitive reference, and retains the compiled validator. Schema compiler
resources use canonical escaped file or private-scheme URLs; loaders decode
their paths once and enforce the configured source boundary. The
provider-facing structured-output metadata uses the root document captured by
the same plan.
Schema preparation and execution remain synchronous. Promptkit checks
cancellation before and after source resolution, JSON decoding, compilation,
and validation, and between bounded schema-read chunks. Once cancellation is
observed it returns the context error without publishing a partial plan or
validation result, even when a compiler or validator has just returned a
different error or a successful result. An `fs.FS` method or JSON Schema
dependency call already in progress cannot be preempted; Promptkit waits for
that call to return and then gives cancellation precedence. Validation does
not detach dependency work into background goroutines.
`Prepare` discards its validation plan after returning metadata. `Run` retains
its plan for initial and repaired-output validation, then discards it with the
operation. `PrepareExecution` retains the plan in its private frozen payload;
`RunPrepared` uses that plan without reopening prompt, profile, input, or
schema sources or rerendering the request. A later ordinary `Run` always
performs fresh source resolution and preparation.
The [validator tests](../../internal/validate/standard_validator_test.go) own The [validator tests](../../internal/validate/standard_validator_test.go) own
basic, JSON, JSON Schema, source resolution, schema loading, compilation, and basic, JSON, JSON Schema, source resolution, schema loading, compilation,
content-failure behavior. frozen-reference behavior, content-failure behavior, and the synchronous
cancellation boundary. Prepared execution
orchestration is owned by the
[use-case tests](../../internal/usecase/prepared_execution_test.go).

View File

@@ -19,24 +19,24 @@ results, public values, extension interfaces, profiles, and error sentinels.
The implemented internal components consist of: The implemented internal components consist of:
- `internal/domain`, which owns framework data values shared by later internal - `internal/domain`, which owns framework data values and source-neutral
components; invariants shared by later internal components;
- `internal/backend`, which owns validated immutable OpenAI-compatible backend - `internal/backend`, which owns validated immutable OpenAI-compatible backend
definitions and the built-in OpenRouter definition; definitions;
- `internal/capacity`, which owns engine-local bounded run admission and - `internal/capacity`, which owns engine-local bounded run admission and
model-generation scheduling for limited backends; model-generation scheduling for limited backends;
- `internal/defaults`, which owns application-neutral framework defaults and - `internal/defaults`, which owns application-neutral framework defaults and
constructs the default execution target; constructs the default execution target;
- `internal/filecatalog`, which discovers YAML files and provides source-path - `internal/filecatalog`, which discovers YAML files and provides source-path
helpers for filesystem and `fs.FS` consumers; helpers for filesystem and `fs.FS` consumers;
- `internal/catalog`, which validates immutable external maintained-catalog
assets before they are eligible for engine assembly;
- `internal/jsonvalue`, which validates and defensively copies JSON-compatible - `internal/jsonvalue`, which validates and defensively copies JSON-compatible
extra-parameter trees; extra-parameter trees;
- `internal/promptdef`, which loads and validates prompt definitions from - `internal/promptdef`, which loads and validates prompt definitions from
filesystem and `fs.FS` sources; filesystem and `fs.FS` sources;
- `internal/profile`, which loads, validates, and overlays execution profiles - `internal/profile`, which loads, validates, and overlays execution profiles
from filesystem and `fs.FS` sources; from filesystem and `fs.FS` sources;
- `internal/profile/builtin`, which embeds the built-in execution profile
catalog;
- `internal/prompt`, which renders prompt messages from Go templates; - `internal/prompt`, which renders prompt messages from Go templates;
- `internal/artifact`, which resolves ordinary inline and unrestricted - `internal/artifact`, which resolves ordinary inline and unrestricted
caller-selected file references; caller-selected file references;
@@ -54,13 +54,15 @@ packages or participate in internal assembly.
The root facade assembles one immutable backend registry, one capacity manager, The root facade assembles one immutable backend registry, one capacity manager,
the internal repositories, renderer, validator, outbound client, and use-case the internal repositories, renderer, validator, outbound client, and use-case
runner while translating public values and errors at the library boundary. The runner while translating public values and errors at the library boundary. The
registry contains built-ins plus validated engine-scoped consumer additions. registry contains validated maintained definitions plus engine-scoped consumer
additions.
The facade constructs the capacity manager from the registry's immutable The facade constructs the capacity manager from the registry's immutable
policy snapshot, wraps the selected built-in or injected model client, and policy snapshot, wraps the selected built-in or injected model client, and
supplies bounded admission to the runner. The defaults and renderer depend on supplies bounded admission to the runner. The defaults and renderer depend on
the domain model. Prompt-definition and profile repositories use the domain the domain model. Prompt-definition and profile repositories use the domain
model, file catalog, and YAML decoder. The built-in profile repository supplies model, file catalog, and YAML decoder. The catalog adapter validates imported
an embedded `fs.FS` to the profile package. Artifact reading uses the domain immutable maintained data before root assembly supplies it to the profile
package. Artifact reading uses the domain
model and application-neutral defaults. Validation uses the domain model, file model and application-neutral defaults. Validation uses the domain model, file
catalog, and JSON Schema implementation. The model client uses the domain catalog, and JSON Schema implementation. The model client uses the domain
model, application-neutral defaults, and an injected or standard-library HTTP model, application-neutral defaults, and an injected or standard-library HTTP
@@ -92,6 +94,13 @@ coordinates internal components and adapts the supported public extension
interfaces to narrow internal abstractions. Internal components must not depend interfaces to narrow internal abstractions. Internal components must not depend
on consumers or on Scriptorium. on consumers or on Scriptorium.
`internal/domain` owns source-neutral invariants for values shared across
multiple input and execution boundaries, including execution-setting bounds,
OpenAI-compatible base endpoints, session identifiers, and output-contract
legality. Callers retain source parsing, required-field rules, other
source-specific normalization, defaulting, error classification, and policy
specific to their own boundary.
## Repository And Consumer Boundary ## Repository And Consumer Boundary
Scriptorium is a downstream application that consumes Promptkit through Scriptorium is a downstream application that consumes Promptkit through

View File

@@ -50,19 +50,18 @@ Examples of appropriate seams include clocks, randomness, subprocesses, remote A
## Test execution requirements ## Test execution requirements
Promptkit currently uses maintainer-run validation rather than hosted CI. Promptkit currently uses maintainer-run validation rather than hosted CI.
Maintainers run the repository-documented test, vet, build, formatting, Maintainers run the complete local workflow in the
documentation-link, and repository-hygiene checks before accepting changes. [development guide](../development.md#maintainer-validation) before accepting
changes. That guide is the canonical owner of exact commands, formatting,
documentation-link validation, and repository-hygiene checks.
Introducing hosted CI later would supplement, not silently redefine, this Introducing hosted CI later would supplement, not silently redefine, this
documented validation model. documented validation model.
The complete test sequence includes ordinary and race-enabled package tests. Maintainer validation must include ordinary and race-enabled package tests,
The maintained offline consumer workflow is also run from the repository root: static analysis, a complete build, and execution of both maintained offline
consumer examples. The preparation example protects assembled preparation and
```sh inspection behavior. The execution example separately protects assembled
go test ./... `Run`, injected-client, validation, usage, and result behavior.
go test -race ./...
go run ./examples/go-library/prepare
```
Tests in the default suite must be deterministic, offline, and independent of Tests in the default suite must be deterministic, offline, and independent of
real credentials. They must not invoke paid APIs, use live network real credentials. They must not invoke paid APIs, use live network
@@ -88,8 +87,9 @@ Use each test type where it protects a distinct risk:
interaction, while replacing live or nondeterministic external boundaries. interaction, while replacing live or nondeterministic external boundaries.
- External-package root tests exercise the public facade as a Go consumer, - External-package root tests exercise the public facade as a Go consumer,
while internal package tests own focused implementation behavior. while internal package tests own focused implementation behavior.
- The maintained offline preparation example protects one representative - The maintained offline preparation and execution examples protect distinct
assembled consumer workflow without contacting a model provider. representative assembled consumer workflows without contacting a model
provider.
- Fixtures should be minimal, synthetic, versioned with the behavior they - Fixtures should be minimal, synthetic, versioned with the behavior they
exercise, and free of credentials or private data. exercise, and free of credentials or private data.
- Golden files are appropriate only when the complete output is intentionally - Golden files are appropriate only when the complete output is intentionally

View File

@@ -106,48 +106,12 @@ gitea.maximumdirect.net/eric/promptkit 1.25.5
promptkit gitea.maximumdirect.net/eric/promptkit promptkit gitea.maximumdirect.net/eric/promptkit
``` ```
Run the complete maintainer validation required by the As a release prerequisite, run the complete
[development guide](development.md): [maintainer validation workflow](development.md#maintainer-validation) against
the clean candidate. Do not substitute a partial command list: the development
```sh guide owns the tests, race checks, analysis, build, both offline examples,
go test ./... formatting, Markdown links, generated-output and credential review, and
go test -race ./... repository hygiene. Record the successful workflow result with the candidate.
go vet ./...
go build ./...
go run ./examples/go-library/prepare
```
Check every tracked Go file. This command must produce no output:
```sh
unformatted=$(
git ls-files '*.go' |
while IFS= read -r go_file
do
gofmt -l "$go_file"
done
)
test -z "$unformatted"
```
Follow every maintained Markdown link and confirm that its local or published
target exists. Review the repository for generated binaries, test or coverage
output, credentials, template residue, downloaded assets, and other files that
do not belong in source control.
Recheck module and repository hygiene, whitespace, and the clean checkout:
```sh
test -z "$(git ls-files go.work go.work.sum)"
test ! -e vendor
if grep -Eq '^[[:space:]]*replace([[:space:]]|\()' go.mod
then
printf '%s\n' 'go.mod contains a replacement' >&2
exit 1
fi
git diff --check
test -z "$(git status --porcelain)"
```
## Write The Release Note ## Write The Release Note
@@ -182,6 +146,13 @@ grep -F 'Consumer action:' "$RELEASE_NOTES_FILE"
Inspect the complete message and confirm that it accurately records the Inspect the complete message and confirm that it accurately records the
compatibility impact, public API changes, and required consumer action. compatibility impact, public API changes, and required consumer action.
When a release introduces or updates the maintained external catalogs, state
that it selects two independently versioned data dependencies, preserves the
public API and configuration, requires no consumer migration, and guarantees
only the catalog versions selected and tested by that Promptkit release. Link
to the current architecture and format documentation rather than duplicating
catalog contracts in the release note.
## Create And Inspect The Tag ## Create And Inspect The Tag
Run the candidate guard again immediately before tag creation. This ensures Run the candidate guard again immediately before tag creation. This ensures

View File

@@ -90,7 +90,7 @@ engine, err := promptkit.NewEngine(
``` ```
See the See the
[custom-backend consumer guide](../consumers/pkg-promptkit.md#register-a-custom-backend) [local-endpoint consumer guide](../consumers/pkg-promptkit.md#configure-a-local-openai-compatible-endpoint)
for task-oriented usage. The for task-oriented usage. The
[`Backend` and `WithBackend` GoDoc](../../backends.go) owns exact registration, [`Backend` and `WithBackend` GoDoc](../../backends.go) owns exact registration,
validation, copying, defaulting, and uniqueness semantics. The validation, copying, defaulting, and uniqueness semantics. The

74
docs/releases/v0.3.0.md Normal file
View File

@@ -0,0 +1,74 @@
# Promptkit v0.3.0
This supplemental changelog summarizes the consumer-facing changes from
`v0.2.0` to `v0.3.0`. The annotated `v0.3.0` tag is the authoritative release
record. Exact current contracts belong to the linked GoDoc and durable
documentation.
## Summary
`v0.3.0` adds a concise way to register the common local OpenAI-compatible
backend configuration:
- `BackendLocal` provides the conventional, non-reserved backend ID `"local"`;
and
- `LocalBackend` constructs an ordinary `Backend` from an endpoint and
concurrency limit.
The helper is explicit and additive. It does not pre-register a backend, read
environment variables, select a model, or replace the complete `Backend`
configuration interface.
## Compatibility
Existing `v0.2.0` consumers require no migration. Endpoint-only profiles,
complete custom `Backend` values, the built-in OpenRouter backend, and existing
registrations using the literal ID `"local"` continue to work unchanged.
## Upgrade
Update the module dependency with:
```sh
go get gitea.maximumdirect.net/eric/promptkit@v0.3.0
go mod tidy
```
Run the consuming project's ordinary tests and race-enabled tests after the
upgrade.
## Configure A Local Backend
Register the convenience value through the existing `WithBackend` option and
select it from one or more profiles:
```go
engine, err := promptkit.NewEngine(
promptkit.Config{PromptDir: "prompts"},
promptkit.WithBackend(
promptkit.LocalBackend("http://localhost:8000/v1", 2),
),
promptkit.WithProfiles(promptkit.Profile{
ID: "local-summary",
BackendID: promptkit.BackendLocal,
Model: "example-model",
}),
)
```
Use an endpoint-only profile when shared backend identity and capacity policy
are unnecessary. Continue to use a complete keyed `Backend` value for custom
IDs, authentication, extra request parameters, explicit queue capacity, or
multiple local endpoints.
See the
[local-endpoint consumer guide](../consumers/pkg-promptkit.md#configure-a-local-openai-compatible-endpoint)
for task-oriented configuration choices. The
[`BackendLocal`, `LocalBackend`, and `WithBackend` GoDoc](../../backends.go)
owns their exact construction, registration, validation, and concurrency
semantics.
## Consumer Action
None. Adopt the convenience constructor when it simplifies local endpoint
configuration.

189
docs/releases/v0.4.0.md Normal file
View File

@@ -0,0 +1,189 @@
# Promptkit v0.4.0
This supplemental changelog and adoption guide summarizes the consumer-facing
changes from `v0.3.0` to `v0.4.0`. The annotated `v0.4.0` tag is the
authoritative release record. Exact current contracts belong to the linked
GoDoc and durable documentation.
## Summary
`v0.4.0` adds four complementary capabilities:
- opaque prepared-execution handles for preparing once, inspecting safe
details, and executing the same frozen snapshot;
- exact prompt-definition inspection without profile resolution or execution;
- exact profile inspection without selecting a prompt or checking credential
availability; and
- structured backend identity on engine admission-capacity rejection.
These APIs let consumers perform more precise preflight work and retain useful
operational context without reproducing Promptkit's internal resolution logic.
## Compatibility
The release is additive for `v0.3.0` consumers. Existing uses of `Prepare`,
`Run`, backend registration, endpoint-only profiles, local-backend helpers,
runtime overrides, public JSON values, and error sentinels continue to work
without migration.
Capacity rejection now returns a structured error while continuing to match
`ErrCapacityExceeded` through `errors.Is`. Error-string wording and direct
sentinel equality were not public contracts.
The new inspection values, capacity error, and prepared-execution handle do not
have stable JSON representations. `PreparedExecution.Details` returns the
existing stable `PreparedRun` value.
## Upgrade
Update the module dependency with:
```sh
go get gitea.maximumdirect.net/eric/promptkit@v0.4.0
go mod tidy
```
Run the consuming project's ordinary and race-enabled tests after upgrading.
No source migration is required.
## Prepare Once And Execute The Same Snapshot
Consumers that need to persist preparation details before generation can now
prepare an opaque, engine-bound execution:
```go
prepared, err := engine.PrepareExecution(ctx, request)
if err != nil {
// Handle preparation failure.
}
defer prepared.Discard()
details := prepared.Details()
// Persist a consumer-selected, appropriately protected preparation record.
result, err := engine.RunPrepared(ctx, prepared)
```
Preparation freezes the selected sources, rendered messages, effective
settings, input content, structured-output metadata, and validation resources
needed by execution. `Details` returns a fresh, caller-owned,
credential-redacted `PreparedRun`.
A handle belongs to its creating engine and permits one execution attempt.
`RunPrepared` consumes that attempt on success and on operational failure.
`Discard` is idempotent and releases an unclaimed handle's execution-only
state. Consumers should discard handles they will not execute, particularly
when a direct request API key may be retained privately until claim or
discard.
Prepared execution does not reserve backend admission during preparation.
Credential availability and backend admission are checked when execution
begins. The execution context is independent of the preparation context.
See the
[prepared-execution consumer guide](../consumers/pkg-promptkit.md#prepare-now-and-execute-the-same-snapshot-later),
the [`PreparedExecution` GoDoc](../../prepared_execution.go), and the
[`Engine.PrepareExecution` and `Engine.RunPrepared` GoDoc](../../engine.go)
for the exact lifecycle, ownership, cancellation, capacity, timing, and
failure contracts.
## Inspect A Prompt
`Engine.InspectPrompt` resolves one prompt ID and optional version through the
engine's configured prompt source:
```go
inspection, err := engine.InspectPrompt(ctx, "report.summary", "")
```
The result includes prompt identity, the opaque prompt hash, declared default
profile ID, declared input metadata, and normalized output contract. It
structurally loads the selected definition and referenced message content but
does not resolve a profile, load schemas or artifacts, render templates,
reserve capacity, or contact a model.
Use inspection for exact configuration checks and metadata discovery. Use
`PrepareExecution` rather than relying on a prior inspection when later
execution must freeze one exact source state, because filesystem-backed
inspection is only a point-in-time lookup.
See the
[prompt-inspection consumer guide](../consumers/pkg-promptkit.md#inspect-a-prompt-before-preparation)
and [`Engine.InspectPrompt` GoDoc](../../engine.go) for exact selection,
ownership, and error behavior.
## Inspect A Profile
`Engine.InspectProfile` resolves one explicit profile independently of a
prompt:
```go
inspection, err := engine.InspectProfile(ctx, "report-production")
```
The result includes the resolved effective execution target and whether a
later request must provide a direct credential. Environment-variable names may
be reported, but inspection does not read credential values or require the
named variable to be populated.
Inspection applies the engine's profile source precedence and resolves any
selected backend. It does not load a prompt, render content, reserve capacity,
or contact a model.
See the
[profile-inspection consumer guide](../consumers/pkg-promptkit.md#inspect-a-profile-before-prompt-work)
and [`Engine.InspectProfile` GoDoc](../../engine.go) for the exact resolution,
credential, ownership, and error contracts.
## Identify Capacity-Rejected Backends
Calls rejected at Promptkit's bounded engine admission boundary continue to
match `ErrCapacityExceeded`. Consumers can additionally obtain the selected
registered backend ID without parsing diagnostic text:
```go
result, err := engine.Run(ctx, request)
if errors.Is(err, promptkit.ErrCapacityExceeded) {
var capacityErr *promptkit.CapacityError
if errors.As(err, &capacityErr) {
// Record capacityErr.BackendID using application-owned diagnostics.
}
// Apply application-owned overload or retry policy.
}
```
The structured error applies to `Run` and `RunPrepared` admission rejection.
It does not represent provider throttling, quota exhaustion, cancellation
while waiting for generation capacity, or another model-client failure.
Promptkit does not prescribe retry timing or transport status mapping.
See the
[error-handling consumer guide](../consumers/pkg-promptkit.md#handle-errors),
the [`CapacityError` GoDoc](../../capacity_error.go), and the
[`ErrCapacityExceeded` GoDoc](../../engine.go) for the canonical contracts.
## Public API Additions
The release adds:
- `Engine.PrepareExecution`;
- `Engine.RunPrepared`;
- `PreparedExecution`, including `Details`, `Discard`, `String`, and
`GoString`;
- `Engine.InspectPrompt`;
- `PromptInspection`;
- `PromptInputDefinition`;
- `Engine.InspectProfile`;
- `ProfileInspection`; and
- `CapacityError`.
No public API was removed.
## Consumer Action
None. Existing `v0.3.0` workflows may upgrade without adopting the new APIs.
Consumers that adopt prepared execution should discard unused handles.
Consumers that need backend-specific capacity diagnostics may add an
`errors.As` check while retaining their existing `errors.Is` classification.

125
docs/releases/v0.5.0.md Normal file
View File

@@ -0,0 +1,125 @@
# Promptkit v0.5.0
This supplemental changelog and migration guide summarizes the consumer-facing
changes from `v0.4.0` to `v0.5.0`. The annotated `v0.5.0` tag is the
authoritative release record. Exact current contracts belong to the linked
GoDoc and durable documentation.
## Summary
`v0.5.0` makes provider requests less prescriptive and adds an application
fallback layer for profile definitions:
- unset optional provider controls are omitted from OpenAI-compatible request
bodies instead of being populated with framework values; and
- `WithFallbackProfileFS` lets an application package profile defaults that
operators can override through the existing ordinary profile sources.
These changes let compatible providers apply their own model defaults while
giving applications stable embedded profile IDs without weakening operator
configuration precedence.
## Compatibility
The release adds one public function and removes no public declaration.
Existing source code should continue to compile.
There is one intentional behavior change: when no profile or runtime override
selects `top_p`, Promptkit no longer sends the former framework value of `1`.
It omits `top_p` and lets the provider choose its behavior. Unset
`temperature` and `max_tokens` are likewise omitted. Explicit nonzero profile
values and runtime values—including explicit runtime zero values—retain their
precedence and wire effect.
Consumers that relied on Promptkit always sending `top_p: 1` should add that
value to the relevant profile or runtime override before upgrading. Consumers
that did not rely on the implicit sampling value require no migration.
Application fallback profiles are opt-in. Engines that do not call
`WithFallbackProfileFS` retain the previous profile-source behavior.
## Upgrade
Update the module dependency with:
```sh
go get gitea.maximumdirect.net/eric/promptkit@v0.5.0
go mod tidy
```
Run the consuming project's ordinary and race-enabled tests after upgrading.
If request payloads or model behavior are asserted in fixtures, review them for
the optional-parameter omission described below.
## Omitted Optional Provider Controls
The built-in OpenAI-compatible client now includes `temperature`,
`max_tokens`, and `top_p` only when a profile or runtime override selects the
value. An explicit runtime zero remains present because runtime override
pointers distinguish zero from an unspecified value.
Promptkit's positive generation deadline remains a framework concern and is
not a provider request-body default. Required request fields, session IDs,
structured output, reasoning selection, credentials, and explicit extra
parameters retain their existing behavior.
See the [framework default and precedence reference](../formats.md#defaults-and-overrides),
the [`ExecutionTargetOverride` GoDoc](../../types.go), and the
[OpenAI-compatible request-body contract](../integrations/openai-compatible-chat.md#request-body)
for current details.
## Embedded Application Fallback Profiles
Applications can package ordinary profile YAML in an `fs.FS` and register it
as a fallback source:
```go
//go:embed profiles/*.yaml
var applicationProfiles embed.FS
engine, err := promptkit.NewEngine(promptkit.Config{
PromptDir: "prompts",
ProfileDir: operatorProfileDir,
},
promptkit.WithFallbackProfileFS(applicationProfiles, "profiles"),
)
```
Leave `operatorProfileDir` empty when no operator source is configured. A
configured ordinary source is authoritative: a matching definition overrides
the application fallback, while a read or validation failure remains an error
instead of silently reaching a lower layer.
Profile definitions resolve in this order:
1. in-memory profiles supplied with `WithProfiles`;
2. the ordinary configured source selected by `WithProfileFile`,
`WithProfileFS`, or `Config.ProfileDir`;
3. the application source supplied with `WithFallbackProfileFS`; and
4. Promptkit's embedded built-in profiles.
Only an absent profile ID falls through. Sources provide complete profiles and
do not merge fields. Loading remains lazy, and the new source uses the existing
strict profile YAML and credential rules.
See the
[embedded-default consumer guidance](../consumers/pkg-promptkit.md#supply-embedded-application-defaults),
the [`WithFallbackProfileFS` GoDoc](../../engine.go), and the
[profile source reference](../formats.md#source-and-profile-precedence) for
current details.
## Public API Changes
The release adds:
- `WithFallbackProfileFS`.
No public declaration was removed or changed.
## Consumer Action
- Review any workflow that depended on Promptkit's implicit `top_p: 1` and
configure the value explicitly when required.
- Optionally adopt `WithFallbackProfileFS` when an application should package
overridable profile defaults.
- Run consumer tests after updating the module dependency.

131
docs/releases/v0.6.0.md Normal file
View File

@@ -0,0 +1,131 @@
# Promptkit v0.6.0
This supplemental changelog and migration guide summarizes the consumer-facing
changes from `v0.5.0` to `v0.6.0`. The annotated `v0.6.0` tag is the
authoritative release record. Exact current contracts belong to the linked
GoDoc and durable documentation.
## Summary
`v0.6.0` is a broad correctness, safety, efficiency, and maintainability
release. It does not add or remove public declarations. The release:
- centralizes shared execution-setting, output-contract, endpoint, and
JSON-compatible-value rules;
- unifies prompt repository behavior and avoids unnecessary prompt and profile
decoding;
- bounds consumer-controlled JSON trees and successful provider responses;
- hardens prompt content paths, artifact files, provider URLs, JSON framing,
and error propagation;
- reuses compiled schema plans and rendered artifact text within an operation;
and
- improves cancellation behavior, prepared-value ownership, deterministic
transport testing, and maintainer validation.
## Compatibility
No public declaration was added, removed, or changed. Ordinary valid `v0.5.0`
configurations and requests should continue to compile and behave as before.
The release intentionally rejects or reports several inputs that were
previously accepted, altered, or misclassified:
- execution settings must be finite, within their documented ranges, and safe
to convert to Go durations;
- output formats, validation modes, repair counts, and JSON Schema dependencies
are validated consistently;
- file-backed prompt and profile identity comes from normalized YAML metadata,
not filenames;
- prompt `content_file` values must be exact relative paths contained by their
configured source root;
- built-in file artifacts must resolve to regular files;
- selected provider endpoints must be absolute HTTP or HTTPS URLs without user
information, query strings, or fragments;
- JSON documents and successful provider responses must contain exactly one
value, and successful provider bodies are limited to 16 MiB; and
- excessively deep or expansive JSON-compatible values fail with ordinary
validation errors.
These are compatibility corrections and safety boundaries rather than new
consumer configuration requirements. Consumers relying on an invalid or
ambiguous input should correct that input before upgrading.
## Upgrade
Update the module dependency with:
```sh
go get gitea.maximumdirect.net/eric/promptkit@v0.6.0
go mod tidy
```
Run the consuming project's ordinary and race-enabled tests after upgrading.
Applications with custom prompt/profile sources, local provider endpoints,
unusual artifact paths, or assertions over provider error identities should
pay particular attention to the compatibility notes below.
## Source Loading And Identity
Prompt definitions now share one source-neutral selection and normalization
flow across operating-system and `fs.FS` sources. YAML `id` and `version`
metadata are authoritative; filenames do not create a second identity system.
Only selected content bodies are loaded, malformed unrelated definitions do
not shadow valid exact matches, and per-file read failures are reported as
prompt-load failures rather than false absence.
File-backed profiles likewise use normalized YAML IDs, reuse their metadata
read for selected strict decoding, and avoid fully decoding unrelated files.
Selected malformed definitions remain authoritative and do not silently fall
through to a lower-precedence source.
Prompt `content_file` paths are opened exactly as declared after a separate
blank check. They must remain relative to and contained by the configured
prompt source root, including across operating-system symlinks.
See the [framework source and identity reference](../formats.md) and
[internal source overview](../internal/sources.md) for the current contracts.
## Validation, Cancellation, And Efficiency
JSON Schema documents preserve exact JSON-number representations. Schema
resource URLs safely escape legal filesystem names, and each operation loads
and compiles its schema graph once. `Run` and prepared execution reuse that
operation-local plan; Promptkit does not introduce a cross-operation cache.
Artifact reading, rendering, schema loading, compilation, and validation now
check cancellation at the synchronous boundaries Promptkit controls. Rendering
memoizes each artifact's text within one render operation, while plain JSON
validation avoids materializing an unnecessary generic tree.
The shared JSON-compatible-value owner now limits nesting and produced work so
unsafe consumer-controlled structures return errors instead of risking
unbounded recursion or allocation. See the
[architecture policy](../policy/architecture.md) for invariant ownership and
the [format reference](../formats.md) for validation behavior.
## Provider Transport Hardening
OpenAI-compatible endpoints are parsed and composed structurally, including
nested base paths. Underlying transport cancellation and deadline errors remain
discoverable with `errors.Is` through Promptkit's generation error category.
Successful provider bodies are read with a fixed 16 MiB bound and must contain
exactly one JSON response object followed only by whitespace. Oversized,
truncated, malformed, or multiply framed responses fail without returning a
partial result. See the
[OpenAI-compatible integration contract](../integrations/openai-compatible-chat.md)
for the canonical request, endpoint, error, and response behavior.
## Public API Changes
None.
## Consumer Action
- Correct any configuration or request that depends on the formerly permissive
cases described under Compatibility.
- Confirm custom local provider endpoints are absolute HTTP or HTTPS base URLs
without credentials, queries, or fragments.
- Confirm prompt content paths remain within their configured source root and
file artifacts resolve to regular files.
- Run ordinary and race-enabled consumer tests after updating the dependency.

155
docs/releases/v0.7.0.md Normal file
View File

@@ -0,0 +1,155 @@
# Promptkit v0.7.0
This supplemental changelog and migration guide summarizes the consumer-facing
changes from `v0.6.0` to `v0.7.0`. The annotated `v0.7.0` tag is the
authoritative release record. Exact current contracts belong to the linked
GoDoc and durable documentation.
## Summary
`v0.7.0` expands provider integration and profile composition while making
credential and generation-failure handling more flexible:
- Promptkit now includes the `rakestrawhome` backend and its Gemma profile;
- built-in generation failures expose bounded structured provider details;
- an unavailable optional API-key environment source no longer prevents a
request from reaching an upstream that permits unauthenticated access; and
- profiles can inherit from and selectively refine another profile.
## Compatibility
This release adds public declarations and fields but removes none. Existing
keyed configuration literals and ordinary `errors.Is` handling continue to
work.
Adding `BaseProfileID` to `Profile` and `OpenAICompatibleProfileConfig` changes
their struct shape. Consumers using positional composite literals for either
type must convert them to keyed literals. Existing keyed literals require no
change.
The `rakestrawhome` backend ID is now built in and reserved. A consumer that
previously registered that exact ID with `WithBackend` must remove its manual
registration before upgrading. Other custom backend registrations are
unchanged.
When an optional backend, profile, or request `APIKeyEnv` is unset, empty, or
whitespace-only, the built-in client now omits `Authorization` and sends the
request. Previously this condition could fail before transport. Set
`Profile.APIKeyRequired` when missing credentials must remain a local
preflight error.
Provider non-success responses continue to match `ErrLLMGenerate`. Their
rendered wording is not a compatibility contract; consumers can now use
`errors.As` with `*GenerationError` when structured status information is
needed.
## Upgrade
Update the module dependency with:
```sh
go get gitea.maximumdirect.net/eric/promptkit@v0.7.0
go mod tidy
```
Remove any manual `rakestrawhome` backend registration, convert positional
profile literals to keyed literals, and run the consuming project's ordinary
and race-enabled tests.
## Rakestrawhome Built-In Backend And Profile
Every engine now includes the reserved `rakestrawhome` backend, identified by
`BackendRakestrawHome`. The built-in `rakestrawhome-gemma-4-31b` profile
selects that backend. Consumers can use the maintained endpoint, credential,
capacity, and model defaults without registering either definition themselves.
See the [built-in backend and profile catalogs](../formats.md#built-in-backends)
and the [consumer adoption example](../consumers/pkg-promptkit.md#use-the-rakestrawhome-built-in-profile)
for the current contracts.
## Structured Generation Errors
Non-2xx responses from the built-in OpenAI-compatible client now return an
immutable `*GenerationError`. Consumers can inspect the HTTP status and any
safely extracted provider code, type, or message while retaining the ordinary
generation-error category:
```go
var generationErr *promptkit.GenerationError
if errors.As(err, &generationErr) {
status := generationErr.StatusCode()
_ = status
}
```
Provider fields are bounded and normalized but remain untrusted and may
contain sensitive request or schema details. Default and Go-syntax formatting
omit those fields. Applications must apply their own disclosure policy before
logging or presenting accessor values.
See the [`GenerationError` GoDoc](../../generation_error.go), the
[consumer error-handling guide](../consumers/pkg-promptkit.md#handle-errors),
and the [OpenAI-compatible response contract](../integrations/openai-compatible-chat.md#response-handling).
## Optional Credential Sources
`APIKeyEnv` names an optional environment lookup source unless the selected
profile explicitly sets `APIKeyRequired`. When neither a direct request key nor
a usable environment value exists, the built-in client omits the bearer header
and handles the upstream response normally. This supports local and other
OpenAI-compatible providers that permit unauthenticated requests without
hiding an authentication error returned by a provider that requires one.
The [credential format reference](../formats.md#credentials), the
[`Backend` GoDoc](../../backends.go), the
[`ExecutionTargetOverride` GoDoc](../../types.go), and the
[authentication integration contract](../integrations/openai-compatible-chat.md#authentication)
define the current precedence and availability rules.
## Profile Inheritance
YAML profiles can name one parent with `base_profile`; in-memory profiles use
`Profile.BaseProfileID`, and `OpenAICompatibleProfileConfig` forwards the same
field. A profile can act as an application-owned alias of a built-in or refine
selected inherited settings:
```go
promptkit.WithProfiles(promptkit.Profile{
ID: "weather-light",
BaseProfileID: "deepseek-4-flash",
ReasoningEffort: "high",
})
```
Base lookup observes the existing source precedence. Chains are linear,
cycle-safe, and resolved afresh for ordinary operations. Prepared execution
freezes the fully resolved target. The selected leaf ID remains public while
effective execution settings reflect the resolved chain.
See the [profile inheritance format reference](../formats.md#profile-inheritance),
the [consumer alias example](../consumers/pkg-promptkit.md#alias-a-built-in-profile),
and the [`Profile` GoDoc](../../types.go) for exact merge and validation
behavior.
## Public API Changes
The release adds:
- `BackendRakestrawHome`;
- `GenerationError`, including `StatusCode`, `ProviderCode`, `ProviderType`,
`ProviderMessage`, `Error`, `GoString`, and `Unwrap`;
- `Profile.BaseProfileID`; and
- `OpenAICompatibleProfileConfig.BaseProfileID`.
No public declaration was removed.
## Consumer Action
- Remove a manual backend registration whose ID is exactly `rakestrawhome`.
- Convert positional `Profile` or `OpenAICompatibleProfileConfig` literals to
keyed literals.
- Set `Profile.APIKeyRequired` where a missing credential must fail locally
instead of reaching the provider unauthenticated.
- Treat `GenerationError` provider fields as untrusted and potentially
sensitive when adopting the new accessors.
- Run consumer ordinary and race-enabled tests after updating the module.

109
docs/releases/v0.8.0.md Normal file
View File

@@ -0,0 +1,109 @@
# Promptkit v0.8.0
This supplemental changelog and migration guide summarizes the consumer-facing
changes from `v0.7.0` to `v0.8.0`. The annotated `v0.8.0` tag is the
authoritative release record. Exact current contracts belong to the linked
GoDoc and durable documentation.
## Summary
`v0.8.0` activates Promptkit's bounded output-repair workflow:
- failed nonempty-text, JSON, and JSON Schema validation can make a limited
number of corrective model calls;
- corrective calls preserve the original rendered conversation, effective
target, session, structured-output contract, and backend capacity policy;
- results report cumulative usage and the number of corrective calls actually
made; and
- explicitly empty OpenAI-compatible response content now reaches output
validation instead of being classified as a malformed provider envelope.
## Compatibility
This release adds no public declarations or fields and removes none. Existing
source code remains source-compatible.
The behavior of the existing `OutputContract.RepairAttempts` field and prompt
YAML `repair_attempts` field has changed. A positive value now authorizes real
additional model calls after eligible validation failures; earlier releases
accepted the field but the public engine remained single-pass. Consumers that
set a positive value should expect additional latency, token usage, and
provider cost when repair is needed.
Repair budgets must now be between zero and three. A positive budget requires
`basic`, `json`, or `json_schema` validation. Values above three and a positive
budget paired with `none` are invalid contracts rather than ignored settings.
An explicitly present empty or whitespace-only string returned by the built-in
OpenAI-compatible client is now a completed generation candidate. `none`
validation permits it, while `basic`, `json`, and `json_schema` classify it
under their ordinary validation rules and may repair it when configured.
Missing, `null`, or non-string content remains a malformed provider response.
## Upgrade
Update the module dependency with:
```sh
go get gitea.maximumdirect.net/eric/promptkit@v0.8.0
go mod tidy
```
Review every prompt definition and request override that sets a positive repair
budget. Use zero or omit the field to retain single-pass execution. Ensure each
positive budget is no greater than three and uses an eligible validation mode,
then run the consuming project's ordinary and race-enabled tests.
## Bounded Output Repair
`repair_attempts` counts corrective calls in addition to the initial model
call. Promptkit validates each completed candidate, stops at the first valid
one, and never exceeds the configured bound. If every candidate remains
invalid, the run completes successfully with the final candidate and its
failed validation result rather than returning an operational error.
Each correction starts from the original rendered messages and includes only
the latest invalid candidate and latest validation diagnostics. JSON Schema
mode retains the provider-native structured-output request as its first line of
defense. Promptkit performs only deterministic structural validation; a valid
response is not necessarily factual or correct for an application's domain.
Usage in the final result is cumulative across the initial response and every
completed corrective response. `ValidationResult.RepairAttempts` reports the
number of corrective calls actually made. Corrective generation failures use
the same public generation-error categories and structured provider details as
an initial generation failure.
See the [output-contract format reference](../formats.md#output-contract), the
[consumer repair example](../consumers/pkg-promptkit.md#repair-a-structured-result),
and the [`OutputContract` and `ValidationResult` GoDoc](../../types.go) for the
current contracts.
## Explicit Empty Content
The built-in OpenAI-compatible client now distinguishes an explicitly present
empty string from a missing or malformed `content` field. This aligns built-in
and injected clients by letting the selected output contract decide whether an
empty candidate is acceptable, invalid, or eligible for repair.
See the
[OpenAI-compatible response contract](../integrations/openai-compatible-chat.md#response-handling)
for the exact envelope behavior.
## Public API Changes
None. This release activates and tightens the documented behavior of existing
fields.
## Consumer Action
- Remove or set `repair_attempts` to zero where execution must remain
single-pass.
- Keep every positive repair budget at three or fewer and pair it with
`basic`, `json`, or `json_schema` validation.
- Account for additional latency, usage, and provider cost when enabling
repair.
- Continue checking the returned validation status because bounded repair can
exhaust without producing a valid candidate.
- Review workflows that previously treated explicit empty provider content as
a generation error.

125
docs/releases/v0.9.0.md Normal file
View File

@@ -0,0 +1,125 @@
# Promptkit v0.9.0
This supplemental changelog and migration guide summarizes the consumer-facing
changes from `v0.8.0` to `v0.9.0`. The annotated `v0.9.0` tag is the
authoritative release record. Exact current contracts belong to the linked
GoDoc and durable documentation.
## Summary
`v0.9.0` adds stateless request-message composition and separates maintained
provider data from Promptkit's core implementation:
- callers can append already-rendered messages to a configured prompt for
application-owned conversations and semantic correction workflows;
- the public package now publishes constants for the four supported text-chat
roles, and prompt definitions use the same normalized role vocabulary; and
- the OpenRouter and Rakestrawhome backend/profile catalogs now come from two
independently versioned Go module dependencies.
## Compatibility
This release adds one field and four constants to the public API and removes no
public declarations. Existing keyed `RunRequest` literals that omit
`AppendedMessages` retain their behavior. Adding the field changes the struct
shape, so consumers using positional `RunRequest` literals must convert them
to keyed literals.
Message roles in prompt definitions are now trimmed, lowercased, and required
to be `developer`, `system`, `user`, or `assistant`. Definitions using another
role that earlier releases accepted as an arbitrary nonblank string now fail
prompt loading. In particular, the text-only message contract does not support
`tool` or the deprecated `function` role. Otherwise valid roles with different
case or surrounding whitespace are normalized rather than rejected.
The catalog extraction preserves Promptkit's public API, built-in backend and
profile IDs, configuration, precedence, credential handling, and capacity
behavior. Consumers do not import or register either catalog themselves, and
no configuration migration is required. Promptkit now selects two
independently versioned data dependencies and guarantees only the catalog
versions selected and tested by this Promptkit release.
## Upgrade
Update the module dependency with:
```sh
go get gitea.maximumdirect.net/eric/promptkit@v0.9.0
go mod tidy
```
Convert any positional `RunRequest` literals to keyed literals. Review prompt
definitions for unsupported roles, then run the consuming project's ordinary
and race-enabled tests.
## Appended Request Messages
`RunRequest.AppendedMessages` accepts caller-owned `RenderedMessage` values
that Promptkit validates, copies, and appends after every rendered
prompt-definition message in caller order. Promptkit does not template this
content, retain conversation state between calls, impose a retry policy, or
apply a message-count, byte-size, token, or context-window limit. Upstream
rejections continue through the ordinary generation-error boundary.
This primitive supports application-owned conversation continuations and
domain-aware correction loops while preserving Promptkit's existing
preparation, hashing, prepared-execution, structural repair, backend-capacity,
credential, and cancellation behavior. Appended content can include sensitive
model output or application feedback; prepared values expose the complete
effective messages by design, while default request formatting reports only
the appended-message count.
See the
[consumer appended-message example](../consumers/pkg-promptkit.md#append-already-rendered-messages),
the [`RunRequest`, `RenderedMessage`, and `CacheControl` GoDoc](../../types.go),
and the [OpenAI-compatible request contract](../integrations/openai-compatible-chat.md#request-body)
for current behavior.
## Supported Message Roles
The new `RoleDeveloper`, `RoleSystem`, `RoleUser`, and `RoleAssistant`
constants identify the complete role vocabulary accepted by Promptkit's
text-chat message model. The same validation and normalization now apply to
prompt-definition messages and request-supplied appended messages. Promptkit
does not translate between roles; provider- or model-specific rejection of an
otherwise supported role remains an upstream generation error.
See the [message format reference](../formats.md#messages-and-templates) for
the canonical prompt-definition contract.
## Independently Versioned Backend Catalogs
Promptkit imports immutable catalog data for its maintained OpenRouter and
Rakestrawhome backends and profiles. Engine construction validates and
assembles both catalogs behind the existing built-in registry and profile
precedence rules. Promptkit no longer keeps duplicate embedded profile assets
or hard-coded definitions for those maintained backends.
The module versions in Promptkit's `go.mod` identify the catalog releases
tested with this release. The [built-in backend and profile format
reference](../formats.md#built-in-backends) remains the canonical consumer
contract, while the [internal source documentation](../internal/sources.md#profiles-and-built-ins)
describes the dependency boundary.
## Public API Changes
The release adds:
- `RunRequest.AppendedMessages`;
- `RoleDeveloper`;
- `RoleSystem`;
- `RoleUser`; and
- `RoleAssistant`.
No public declaration was removed.
## Consumer Action
- Convert positional `RunRequest` literals to keyed literals.
- Replace unsupported prompt-definition roles with an appropriate supported
role, or keep richer tool-call protocols in an application-owned client.
- Treat appended messages and prepared effective messages according to the
application's sensitive-data policy.
- Do not add direct catalog imports or registration calls; existing Promptkit
construction and configuration remain correct.
- Run consumer ordinary and race-enabled tests after updating the module.

View File

@@ -1,236 +0,0 @@
# Backend-Specific Concurrency Management
**Status:** Complete.
## Purpose
This roadmap defines the scope and target end state for engine-local,
backend-specific concurrency management. It records the intended capability,
consumer value, and important policy choices.
This document is planning material, not a description of current behavior.
Current exported contracts remain owned by Go declarations and GoDoc, backend
registration guidance by the
[consumer guide](../consumers/pkg-promptkit.md#register-a-custom-backend), and
implemented orchestration by the
[internal runner document](../internal/runner.md).
## Motivation
Different model backends can sustain very different request loads. A local
network endpoint may need a small concurrency limit, while OpenRouter can
usually accept substantially more simultaneous work. Requiring every consumer
to build its own semaphores and queues would duplicate routing knowledge,
create inconsistent cancellation behavior, and make it easy for one caller to
bypass the intended backend limit.
Promptkit should own this coordination because it already resolves each run to
an engine-scoped backend identity and owns every model-generation call made by
the runner. Consumers should continue submitting ready-to-run requests through
the synchronous API, including concurrently from multiple goroutines, without
implementing their own backend scheduler.
The buffered queue is a safety boundary, not an ordinary throughput
restriction. Its primary purpose is to prevent a bug or unintended submission
loop from creating an unbounded in-memory backlog.
## Scope
The feature will add optional concurrency policy to registered backends and
coordinate `Run` calls against independent per-backend capacity pools.
Each policy has two distinct controls:
- an active-generation limit, which protects the backend from too many
simultaneous model requests; and
- a bounded waiting capacity, which protects the process from admitting an
unbounded backlog.
Concurrency policy belongs to a backend registration. It is not a profile
model parameter and cannot be overridden per run. Profiles select the policy
through their backend ID, while a profile or request endpoint override remains
in the selected backend's pool.
`Prepare` does not call a model and will remain outside concurrency admission.
## Defaults And Configuration
The built-in OpenRouter backend will use:
- an active-generation limit of 16; and
- a waiting capacity of 1024.
The waiting default is intentionally generous. Reaching it should indicate
abnormal submission pressure rather than normal application behavior.
Consumer-registered backends will remain unlimited unless the consumer
configures an active-generation limit. When a consumer enables a limit and
does not specify waiting capacity, the waiting capacity will default to 1024.
Consumers may configure a different bounded capacity, including zero when
they want no admitted backlog beyond the active-limit-sized run set.
The public representation must distinguish an omitted waiting capacity from
an explicit zero.
Endpoint-only profiles have no backend registration from which to obtain
policy and will remain unlimited. A future engine-wide or endpoint-keyed
policy can be considered separately if consumers demonstrate that need.
Invalid limits or capacities will fail engine construction as invalid
configuration. Policy values will be copied into engine-owned immutable state
along with the rest of the backend registration.
## Admission And Execution Behavior
`Run` remains a synchronous, wait-for-result operation. Concurrent callers may
block inside `Run` while waiting for their selected backend, then receive the
ordinary result or error from that invocation.
For a configured pool, the active-generation limit plus the waiting capacity
defines the maximum number of concurrent `Run` invocations that Promptkit will
accept for that backend. A waiting capacity of zero therefore accepts no more
runs than the active limit. Admission is immediate: a call either reserves one
of those bounded slots or receives the capacity error. An accepted run may
then wait internally for active-generation capacity.
For a limited backend, Promptkit will bound the number of accepted runs before
expensive artifact loading, prompt rendering, and large defensive copies where
practical. Lightweight prompt, profile, and backend resolution may occur first
when it is required to identify the correct capacity pool. This pre-admission
resolution must not become a second execution-precedence path with behavior
that can drift from `Prepare`.
An accepted run retains its admission until it completes or fails. Every
actual model-generation call for that run must separately observe the
backend's active-generation limit. This includes:
- the initial generation;
- every output-repair generation; and
- calls made through either the built-in or an injected model client.
Preparation and output validation should not hold an active-generation permit.
A repair remains part of its already-admitted run, but reacquires active
generation capacity so repairs cannot exceed the backend limit. It must not be
rejected merely because new runs filled the waiting queue after its initial
generation.
Within one backend pool, waiting generation calls should be served in FIFO
order, subject to canceled calls being removed. Different backend pools make
progress independently; a saturated local backend must not consume
OpenRouter's active or waiting capacity.
The feature will not promise ordering across backend pools or completion order
among admitted runs.
## Capacity Failure And Cancellation
When a backend's bounded waiting capacity is full, a new `Run` call will fail
promptly rather than waiting outside the bounded admission system. The public
API will expose a recognizable capacity-exhaustion error identity distinct
from invalid configuration, invalid requests, and model-client failures.
Rejected calls return no partial result and do not invoke the model client.
Waiting within the admitted backlog or for active-generation capacity must
honor the caller's context. Cancellation or deadline expiry will:
- stop waiting promptly;
- release any admission or generation capacity held by that invocation;
- preserve the applicable context error identity; and
- avoid invoking the model client if cancellation wins before generation
starts.
Capacity must also be released after preparation, generation, validation,
repair, or collaborator failure. One failed or canceled run must not reduce
the backend's future usable capacity.
Elapsed `Run` timing will include time spent waiting after the call is
accepted. `PreparedRun` timing will continue to describe preparation rather
than queue waiting.
## Engine And Client Boundaries
All pools and queued state belong to one `Engine`. Separate engines do not
share capacity, even when they register the same backend ID or endpoint. The
feature introduces no process-global scheduler.
The engine will apply policy consistently to the built-in model client and an
injected `LLMClient`. Consumers calling their own client outside Promptkit are
outside this boundary. Injected clients remain responsible for their internal
thread safety and cancellation behavior.
Backend policy is keyed by the resolved backend ID rather than endpoint text.
This preserves stable routing when a selected backend's endpoint is overridden
and avoids accidentally combining unrelated registrations that happen to use
the same URL.
## Queue Lifetime And Observability
Admission state is buffered, ephemeral, and in-process. It is not persisted
and has no survival guarantee across engine disposal or process termination.
Promptkit will not introduce background job ownership or require consumers to
start or stop workers.
The initial feature does not require public queue-depth metrics, callbacks, or
inspection APIs. Capacity errors and ordinary call timing provide the
consumer-visible behavior. Operational observability can be added later
without coupling the scheduling mechanism to an application logging or
metrics system.
## Compatibility
Consumer-registered backends and endpoint-only profiles remain unlimited
unless concurrency is explicitly configured, preserving their existing
behavior.
The built-in OpenRouter backend will change from unlimited concurrency to a
limit of 16 with a bounded waiting capacity of 1024. Ordinary synchronous
calls remain unchanged, while unusually high concurrent use may now wait or
return the capacity error. This behavioral change must be identified in the
release notes for the version that publishes it.
Adding backend policy fields and a public capacity error is otherwise
additive. The change will use a pre-`v1` minor release under Promptkit's
[release policy](../release.md#release-model).
## Non-Goals
This scope does not include:
- asynchronous job handles, polling, or detached result delivery;
- durable or cross-process queues;
- persistence or recovery across engine or process shutdown;
- priorities, scheduling weights, or consumer-defined fairness classes;
- automatic retries, backoff, rate-limit interpretation, or provider quota
discovery;
- token-per-minute or request-per-minute rate limiting;
- dynamic reconfiguration after engine construction;
- per-profile or per-run concurrency overrides;
- endpoint-keyed pooling for profiles without a backend ID;
- process-global coordination across engines;
- application worker lifecycle, logging, tracing, or metrics policy; or
- changes to prompt, profile, schema, or model-provider wire formats.
## Target End State
This roadmap reaches its target end state when:
- each engine independently coordinates configured backend capacity;
- the built-in OpenRouter backend allows 16 active generations and up to 1024
waiting runs;
- consumer backends can opt into their own active and waiting limits while
remaining unlimited by default;
- endpoint overrides retain the selected backend's capacity pool and
endpoint-only profiles remain unlimited;
- synchronous `Run` callers wait for and receive their ordinary result;
- admission is bounded before expensive preparation work where practical;
- every initial and repair generation observes the backend's active limit
without serializing preparation or validation;
- a full waiting queue returns a recognizable capacity error without invoking
the model client;
- cancellation and all failure paths promptly release capacity and preserve
context error identity;
- built-in and injected model clients receive the same scheduling behavior;
- pools remain ephemeral, engine-scoped, and independent across backend IDs;
and
- current-state GoDoc, consumer, internal, and release documentation describe
the implemented behavior once it lands.

61
docs/roadmap/deferred.md Normal file
View File

@@ -0,0 +1,61 @@
# Deferred Feature Ideas
## Purpose
This document catalogs feature ideas that remain potentially useful but have
been deliberately postponed. These ideas are not awaiting ordinary selection
from the [future feature catalog](future.md); each has a stated reason to wait
and should be reconsidered only when its trigger becomes relevant.
Deferred entries are not commitments, schedules, active implementation plans,
or descriptions of current behavior. When an entry is reactivated, move it to
`future.md` for evaluation or directly into a focused roadmap after its open
design dependencies have been resolved.
## Deferred Ideas
### Semantic Execution-Target Fingerprints
**Reason for deferral:** A stable digest requires a deliberate semantic-
equality and versioning design. Notarius can safely use conservative source
hashes and a Promptkit release marker today, while Weatherreporter does not
currently reuse LLM-dependent checkpoints.
Promptkit could expose an opaque equality value for a resolved profile and its
effective generation target. This would let checkpointing consumers detect
generation-affecting configuration changes without hashing YAML presentation
or depending on Promptkit's built-in catalog layout.
The digest should change with semantically relevant state such as the resolved
model, endpoint, backend routing identity, request defaults, extra parameters,
profile generation settings, and selected built-in profile semantics. It
should exclude credential values, concurrency and queue policy, source paths,
comments, formatting, and other representation-only changes. Whether a
credential environment-variable name affects equality must be decided
explicitly. The encoding should remain opaque and internally versioned so
Promptkit can deliberately invalidate earlier digests when its resolution
semantics change.
Reconsider this idea when a downstream consumer needs Promptkit-owned
checkpoint equality or when a broader semantic identity design is selected.
### Eager Source Validation
**Reason for deferral:** Exact prompt and profile inspection may already
provide a sufficiently small validation surface. Experience from downstream
adoption should establish whether an engine-wide operation would add enough
value to justify its broader contract.
Promptkit could provide an explicit offline operation that discovers and
structurally validates configured prompt, profile, and schema sources without
model generation. The normal `NewEngine` path would remain lazy.
An eager operation would need coherent handling for duplicate prompt IDs and
versions, strict YAML decoding, referenced content files, profile/backend
membership, schema syntax and transitive references, context cancellation,
and source-specific public errors. Credential declarations must remain
separate from credential values; checking current environment availability,
if supported at all, should be an explicit option and must not expose secrets.
Reconsider this idea after downstream use of `InspectPrompt`,
`InspectProfile`, and fixture-based preparation demonstrates a concrete gap.

View File

@@ -12,6 +12,9 @@ consumer value, and important scope boundaries. Defer API design,
implementation details, sequencing, and acceptance criteria until an idea is implementation details, sequencing, and acceptance criteria until an idea is
selected. selected.
Ideas that have been deliberately postponed rather than left available for
ordinary selection belong in the [deferred catalog](deferred.md).
## Using This Catalog ## Using This Catalog
- Add an idea when its purpose and likely value can be stated clearly. - Add an idea when its purpose and likely value can be stated clearly.
@@ -23,6 +26,8 @@ selected.
- When an idea is selected, move its active planning to a focused roadmap or, - When an idea is selected, move its active planning to a focused roadmap or,
when it requires a durable architectural decision, an ADR. Update when it requires a durable architectural decision, an ADR. Update
current-state documentation only when implementation lands. current-state documentation only when implementation lands.
- Move an idea to `deferred.md` when maintainers decide to retain it but wait
for a stated design dependency, demand signal, or reconsideration trigger.
- Remove ideas that are no longer relevant. Retain a rejected idea only when - Remove ideas that are no longer relevant. Retain a rejected idea only when
its rationale is likely to prevent repeated reconsideration. its rationale is likely to prevent repeated reconsideration.
@@ -33,9 +38,8 @@ consumers.
## Ideas ## Ideas
No ideas are currently cataloged. Backend-specific concurrency management has No ideas are currently awaiting selection. Active feature work belongs in its
been selected for active planning in the focused roadmap rather than this catalog.
[focused concurrency roadmap](concurrency.md).
## Entry Format ## Entry Format

View File

@@ -1,841 +0,0 @@
# Backend-Specific Concurrency Management Implementation Plan
**Status:** Complete.
## Purpose
This document is the decision-complete implementation plan for
[backend-specific concurrency management](concurrency.md). It is written for a
coding agent that will implement each stage in order.
The feature roadmap owns the intended capability, consumer value, policy
choices, compatibility decision, and target end state. This document owns the
concrete API, internal representation, scheduling architecture, implementation
sequence, test ownership, documentation updates, and completion gates.
## Implementation Rules
- Complete the stages in order. Keep the repository compiling and the focused
tests passing at every stage boundary.
- Preserve unrelated working-tree changes. In particular, `concurrency.md` and
the removal of its source idea from `future.md` may already be uncommitted
when implementation begins; retain both.
- Follow every policy under `docs/policy/`, the task-specific reading guide in
`docs/development.md`, and the target behavior in `concurrency.md`.
- Keep the public API in the root `promptkit` package and implementation
details under `internal/`. Do not expose scheduler types or create another
public package.
- Use only the Go standard library for scheduling. Do not add a queue,
semaphore, worker-pool, or metrics dependency.
- Preserve synchronous, wait-for-result `Run`, unrestricted `Prepare`,
engine-local state, endpoint-only profiles, backend-selected profiles,
backend identity through endpoint overrides, and injected `LLMClient`
behavior.
- Do not broaden the work into asynchronous jobs, durable queues, retries,
rate limiting, dynamic configuration, priorities, worker lifecycle,
endpoint-keyed pools, or public queue observability.
- Keep all tests deterministic, bounded, offline, and race-safe. Coordinate
concurrent tests with channels and barriers rather than timing assumptions
or live providers.
- Update exact GoDoc with each exported declaration change. Update durable
current-state documents only after the corresponding behavior is
implemented.
- Test configurable mechanisms with small test-owned limits. Assert the exact
OpenRouter `16` and default queue `1024` values only at the registry contract
that owns those operational defaults.
- Do not create a release, change a module version, or tag a commit. The final
implementation handoff must identify the built-in OpenRouter behavior change
for the next pre-`v1` minor release.
## Fixed Design
### Public Backend Configuration
Append these fields to the existing root `Backend` type in `backends.go`:
```go
type Backend struct {
// Existing fields remain unchanged and in their current order.
ConcurrencyLimit int
QueueCapacity *int
}
```
Use these exact semantics:
| Public values | Meaning |
| --- | --- |
| `ConcurrencyLimit == 0`, `QueueCapacity == nil` | Unlimited backend; preserve current behavior. |
| `ConcurrencyLimit > 0`, `QueueCapacity == nil` | Limit active generations and use the default waiting capacity of 1024. |
| `ConcurrencyLimit > 0`, `QueueCapacity != nil` | Limit active generations and use the pointed-to capacity exactly, including zero. |
| `ConcurrencyLimit < 0` | Invalid engine configuration. |
| `QueueCapacity != nil` and `*QueueCapacity < 0` | Invalid engine configuration. |
| `ConcurrencyLimit == 0` and `QueueCapacity != nil` | Invalid engine configuration because a queue without an active limit has no defined consumer value. |
`ConcurrencyLimit` counts simultaneous calls to the engine-owned internal
model-client boundary for this backend. `QueueCapacity` controls additional
accepted `Run` invocations beyond that limit. The maximum admitted runs for a
limited backend is therefore:
```text
ConcurrencyLimit + effective QueueCapacity
```
Guard that addition against integer overflow during backend validation.
Do not impose an arbitrary upper bound beyond non-negativity and overflow
safety.
The `QueueCapacity` pointer exists only to distinguish omission from explicit
zero. `WithBackend` and `NewEngine` must not retain the caller's pointer.
`Backend` continues to have no stable JSON representation, and consumers
remain directed to keyed literals.
Do not add concurrency fields to `Profile`, `ExecutionTarget`,
`ExecutionTargetOverride`, `RunRequest`, prompt or profile files, or stable
prepared/result JSON.
### Built-In And Custom Defaults
The backend registry owns these exact operational defaults:
```go
const (
openRouterConcurrencyLimit = 16
defaultQueueCapacity = 1024
)
```
The built-in `openrouter` definition has a normalized concurrency limit of 16
and queue capacity of 1024.
Consumer registrations remain unlimited when concurrency is omitted. For a
consumer backend with a positive limit and omitted queue capacity, normalize
the queue capacity to 1024. Preserve an explicitly configured zero.
Consumers still cannot replace the reserved `openrouter` registration.
Endpoint-only profiles have no backend policy and remain unlimited. A selected
backend retains its pool when a profile or request overrides only its endpoint.
### Internal Backend Representation
Extend `internal/domain.Backend` with scalar policy values and explicit
presence rather than retaining a pointer:
```go
type Backend struct {
// Existing fields...
ConcurrencyLimit int
QueueCapacity int
QueueCapacitySet bool
}
type BackendCapacityPolicy struct {
ConcurrencyLimit int
QueueCapacity int
}
```
`WithBackend` converts the public pointer into `QueueCapacity` plus
`QueueCapacitySet`. Registry normalization validates the combinations above,
fills the default, and leaves every limited stored backend with
`QueueCapacitySet == true`. Unlimited stored backends retain zero values and
`QueueCapacitySet == false`.
Add this internal registry method:
```go
func (r *Registry) CapacityPolicies() map[string]domain.BackendCapacityPolicy
```
It returns a newly allocated map containing only limited backends. Values are
scalars, so callers cannot mutate registry state. The built-in OpenRouter
policy is included. `GetBackend` continues returning a defensive backend copy,
now including normalized scalar capacity metadata.
Capacity policy is operational registry metadata. Do not merge it into an
execution target or expose it to injected model clients.
### Public Capacity Error
Add this root sentinel beside the other run errors in `engine.go`:
```go
var ErrCapacityExceeded = errors.New("backend capacity exceeded")
```
Its GoDoc must state that it identifies a `Run` rejected because the selected
backend has already admitted `ConcurrencyLimit + QueueCapacity` runs. It is
not an invalid request, an LLM/provider rate-limit response, or an
`ErrLLMGenerate` failure.
The internal capacity component owns a corresponding internal
`ErrCapacityExceeded`. Add its mapping in `publicErrorFor` before the broader
generation and invalid-request cases. The public error must preserve the
internal error through wrapping while matching `ErrCapacityExceeded` with
`errors.Is`.
A capacity rejection returns no partial result and must not invoke the
artifact reader, renderer, schema loader, validator, or model client. Prompt,
profile, and backend loading needed to select the pool may already have
occurred.
### Internal Capacity Component
Add `internal/capacity` as the single owner of engine-local run admission and
active-generation permits.
Use these package-level boundaries:
```go
var ErrCapacityExceeded error
type Manager struct {
// Private immutable pool map.
}
func NewManager(
policies map[string]domain.BackendCapacityPolicy,
) (*Manager, error)
func (m *Manager) Admit(
ctx context.Context,
backendID string,
) (release func(), err error)
func NewClient(m *Manager, next llm.Client) llm.Client
```
`NewManager` copies the supplied map and creates one independent pool per
limited backend. Defensively reject blank IDs, non-positive concurrency
limits, negative queue capacities, or total-capacity overflow even though the
registry normally supplies normalized values. Construction creates no worker
goroutines.
An absent manager, blank backend ID, or ID absent from the policy map is
unlimited:
- `Admit` succeeds with a non-nil no-op release function; and
- the client wrapper calls the next client directly.
For a limited pool, `Admit` is immediate and context-aware:
1. return `ctx.Err()` if the context is already done;
2. under the pool lock, compare admitted runs with
`ConcurrencyLimit + QueueCapacity`;
3. return an error matching internal `ErrCapacityExceeded` when full; or
4. increment admitted runs and return an idempotent release function.
The release function decrements admission exactly once, even if accidentally
called more than once. It does not release an active-generation permit; those
permits have their own lifetime.
### FIFO Generation Permits
`NewClient` returns an internal `llm.Client` wrapper around either the built-in
client or the public-client adapter. It must preserve requests, successful
responses, nil responses, and collaborator error identities exactly.
`next` must be non-nil; `NewEngine` and internal runner construction maintain
that invariant. A nil manager returns `next` unchanged.
For a configured backend ID, the wrapper:
1. acquires one active-generation permit from the matching pool;
2. waits in FIFO order when the active count equals `ConcurrencyLimit`;
3. removes a canceled waiter and returns `ctx.Err()` when cancellation wins
before the permit is granted;
4. invokes the next client only after a permit is granted; and
5. releases the permit with `defer` after every success, nil response,
collaborator error, panic unwinding, or context outcome.
Implement FIFO and cancellation explicitly with a mutex and an ordered waiter
list. A channel used only as a counting semaphore is insufficient because it
does not define FIFO ordering or safe removal of canceled waiters.
Permit grant and cancellation must have one lock-protected linearization
point. If cancellation removes the waiter first, do not invoke the next
client. If grant wins first, invoke the next client with the caller's context;
the next client may then observe cancellation normally. Never lose or
double-release a permit in this race.
Releasing a permit transfers it to the oldest non-canceled waiter before
making it generally available. Different backend pools never share admission
or active counts.
The active wrapper enforces its limit even if an internal caller invokes it
without a run admission lease. Bounded backlog is guaranteed for ordinary
engine `Run` calls by the runner admission path; no public API exposes the
wrapped internal client directly.
### Engine Assembly
In `NewEngine`, after constructing the validated backend registry:
1. obtain `backendRegistry.CapacityPolicies()`;
2. construct one `capacity.Manager`;
3. construct the selected base internal LLM client exactly as today;
4. wrap that base client with `capacity.NewClient`; and
5. pass both the wrapped client and manager-as-admitter to the runner.
Every `NewEngine` call constructs a distinct manager. Do not cache managers,
pools, or policies in package globals. The wrapper must be applied after a
public injected client is adapted to `internal/llm.Client`, so built-in and
injected clients receive identical scheduling behavior.
If `NewManager` reports a defensive configuration error, make `NewEngine`
return an error matching `ErrInvalidConfig`.
`Prepare` does not use the manager. An injected client remains required to be
safe for concurrent calls because different backend pools and unlimited
backends may still invoke it concurrently.
### Shared Two-Phase Preparation
Refactor `internal/usecase.Runner` so `Prepare` and `Run` share one preparation
pipeline with two private phases. Do not duplicate prompt/profile/backend
selection or execution precedence.
The first phase resolves only the state required before admission:
1. validate `PromptID`;
2. normalize the direct session ID;
3. load the prompt definition;
4. hash the original prompt definition at its existing error-order position;
5. select and load the execution profile;
6. resolve the selected backend;
7. resolve and validate the effective execution target and credentials; and
8. resolve the effective output contract without loading its schema.
Return a private state value containing the loaded definition, normalized
direct session, prompt-definition hash, selected profile ID, effective target,
numeric-presence metadata, effective output contract, and preparation start
time. Keep this value private to `internal/usecase`.
The second phase consumes that state and performs:
1. structured-output schema loading;
2. artifact loading and input hashing;
3. message and prompt-session rendering;
4. direct-session application;
5. rendered-prompt hashing; and
6. `PreparedRun` construction and timing.
Preserve every existing precedence rule, error identity, direct-session
template bypass, hash input, selected identity, copy guarantee, and timing
field. Do not reload the prompt, profile, or backend between phases.
`Runner.Prepare` records its start time, runs both phases consecutively, and
never calls admission. Its behavior and error ordering remain unchanged.
`Runner.Run` records its existing run start time, runs the first preparation
phase, and then calls:
```go
release, err := r.admitter.Admit(ctx, effectiveBackendID)
```
Use a narrow use-case-owned interface with the same signature:
```go
type RunAdmitter interface {
Admit(context.Context, string) (func(), error)
}
```
A nil admitter means unlimited behavior for internal constructors and tests.
On successful admission, immediately `defer release()` around the remainder of
the run. Then run the second preparation phase, initial generation,
validation, and all repair attempts.
If admission returns internal `capacity.ErrCapacityExceeded`, add useful
backend context without changing its identity. If it returns `ctx.Err()`,
preserve that identity directly rather than recategorizing it as invalid
request or generation failure.
This refactor intentionally replaces the current literal `Run`-calls-`Prepare`
implementation with shared private phases. Update current-state documentation
to describe one shared pipeline rather than retaining an inaccurate call-graph
claim.
### Generation And Repair Lifetime
The admission lease covers the entire accepted run:
- second-phase preparation;
- initial generation;
- validation;
- every repair; and
- all failure and cancellation exits.
Preparation and validation do not hold an active-generation permit. The
wrapped client acquires a permit only around each actual `Generate` call.
The runner's initial generation already carries the effective backend ID in
`GenerateRequest.Target`. Preserve that value. `RepairRequest.Target` and the
default repairer's generated request must continue carrying the same backend
ID, allowing each repair to reacquire the same pool's active permit.
When testing or constructing `NewRunnerWithRepairer`, pass the same wrapped
client to both the runner and `NewDefaultOutputRepairer`. Do not add capacity
state to `RepairRequest`, `ExecutionTarget`, or public generation values.
A repair remains within its existing admission lease. It waits for a FIFO
active permit but never performs a second bounded admission and therefore
cannot fail merely because later runs filled the admission capacity.
### Error And Cancellation Semantics
The required public outcomes are:
| Situation | Required error identity |
| --- | --- |
| Admission capacity is full | `ErrCapacityExceeded` only; not `ErrInvalidRequest` or `ErrLLMGenerate`. |
| Context is done before admission succeeds | Preserve `ctx.Err()`; do not return capacity exhaustion. |
| Context cancels while waiting for an active permit | Preserve `ctx.Err()` through the existing `ErrLLMGenerate` generation category. |
| Wrapped client fails after permit acquisition | Preserve existing `ErrLLMGenerate` and collaborator identities. |
| Preparation or validation fails after admission | Preserve its existing category and release admission. |
Maintain the existing rule that `Run` returns no partial result on any
operational error. Do not add queue status to errors or results.
`RunResult.Duration` continues to start at runner entry and therefore includes
pre-admission resolution, accepted preparation, and active-permit waiting.
`PreparedRun.DurationMS` continues to cover only its shared preparation phases;
it does not include later generation waiting. Capacity-rejected calls have no
result or timing value.
### Ownership And Concurrency Safety
The registry, capacity policy map, pool map, and per-pool limits are immutable
after engine construction. Only admission counts, active counts, and waiter
lists are mutable and must be protected by the owning pool mutex.
Do not retain public queue pointers, caller request values, contexts, or
generation requests after their call completes. A canceled waiter must be
unlinked so its context and request cannot remain reachable from the pool.
Do not hold a pool mutex while:
- loading or rendering prompts;
- reading artifacts or schemas;
- invoking a model client;
- validating output;
- closing a waiter notification channel if the implementation could re-enter
pool code; or
- calling consumer code.
No scheduler operation may spawn a goroutine whose lifetime outlasts the
calling `Run`. The zero steady-state goroutine count is part of the
in-process/no-worker-lifecycle design.
## Test Ownership
Use this ownership split and avoid repeating the full policy matrix at every
layer:
- `internal/backend/registry_test.go` owns normalization, validation, the exact
OpenRouter policy, the custom default queue, explicit zero, unlimited
omission, and policy-map copying.
- `internal/capacity/manager_test.go` owns admission bounds, idempotent release,
FIFO active permits, cancellation races, capacity recovery, independent
pools, unlimited IDs, and observed peak concurrency.
- `internal/capacity/client_test.go` owns wrapper request/response/error
transparency and the rule that cancellation before grant does not invoke the
next client. Combine these with manager tests if one coherent package test
expresses the behavior more clearly.
- `internal/usecase/runner_test.go` owns two-phase preparation parity, pool
selection, admission before expensive work, admission release across run
exits, `Prepare` bypass, and repair reuse of the admitted backend.
- Root external-package tests own public configuration conversion, assembled
engine-local behavior, endpoint-override routing, injected-client limiting,
and public capacity/context error identities.
- Existing model-client HTTP tests remain unchanged because scheduling does
not alter the OpenAI-compatible wire contract.
Concurrency tests must use test-owned limits such as one or two and
channel-controlled blocking clients. Record observed active and peak counts
under a mutex or atomics. Do not use `time.Sleep` to infer queue state.
Package-internal tests may inspect a waiter list under its mutex through a
small test helper when necessary to establish deterministic FIFO ordering; do
not add production metrics or hooks solely for tests.
Do not add separate tests for trivial scalar copies when registry or assembled
behavior already protects them.
## Stage 1 — Backend Policy And Public Configuration
**Status:** Complete.
### Goal
Add the public and internal backend policy representation, normalize all
configured states, and expose immutable normalized policies without changing
runtime scheduling yet.
### Work
1. Add `ConcurrencyLimit` and `QueueCapacity` to `Backend` in `backends.go`
with exact GoDoc for unlimited, defaulted, explicit-zero, invalid, and
engine-scoped behavior.
2. Convert the public queue pointer into scalar value plus presence in
`WithBackend`; do not retain the pointer.
3. Add the internal backend policy fields and
`BackendCapacityPolicy` to `internal/domain/domain.go`.
4. Add the two registry-owned constants and configure the built-in OpenRouter
definition with 16 and 1024.
5. Extend `normalizeBackend` with the fixed validation, defaulting, explicit
zero, and overflow rules.
6. Add `Registry.CapacityPolicies`, returning only limited policies in a fresh
map.
7. Update existing backend composite literals and assertions only where the
new fields are relevant. Continue using keyed literals.
### Tests
1. Extend the exact built-in registry test with the OpenRouter limit and queue.
2. Add one coherent table covering unlimited omission, default queue,
explicit-zero queue, negative values, queue-without-limit, and total
overflow.
3. Extend the registry copy/normalization test to prove returned policy maps
cannot mutate registry state.
4. Add root coverage only if needed to prove the public pointer/presence
conversion; do not reproduce registry validation cases at the facade.
### Focused Validation
Run:
```sh
gofmt -w backends.go internal/domain/domain.go \
internal/backend/registry.go internal/backend/registry_test.go
go test . ./internal/backend
go vet . ./internal/backend
git diff --check
```
Include another touched Go test file in `gofmt` only if it actually changed.
### Completion Gate
This stage is complete when every public configuration state has one normalized
internal meaning, OpenRouter exposes exactly 16/1024, custom backends remain
unlimited by omission, and no runtime call is scheduled yet.
## Stage 2 — Engine-Local Capacity Manager
**Status:** Complete.
### Goal
Implement and prove the bounded admission mechanism and FIFO active-generation
client wrapper independently of runner orchestration.
### Work
1. Add `internal/capacity/manager.go` with the manager, immutable policy copy,
per-backend pools, internal error, immediate admission, idempotent release,
and FIFO context-aware active permits.
2. Add `internal/capacity/client.go` with the transparent `llm.Client` wrapper.
3. Use mutex-protected waiter state and an ordered list; explicitly resolve
grant-versus-cancel races.
4. Ensure unlimited and independent-pool fast paths avoid queue allocation.
5. Do not start workers, timers, cleanup goroutines, or process-global state.
### Tests
1. Add a compact constructor-validation table for blank IDs, non-positive
limits, negative queues, and total-capacity overflow.
2. With a small configured policy, prove that exactly
`limit + queueCapacity` admissions succeed, the next matches
`ErrCapacityExceeded`, and a release permits another admission.
3. Prove release is idempotent.
4. Drive more blocked client calls than the active limit and assert observed
peak concurrency never exceeds that limit.
5. Prove FIFO order with deterministic queue-entry synchronization.
6. Cancel the first and a middle waiter and prove they are removed, never call
the wrapped client, and do not block later waiters.
7. Exercise the grant/cancel race repeatedly under `go test -race`, asserting
no permit leak or double invocation.
8. Prove different backend IDs proceed independently and blank, unknown, or
nil-manager paths remain unlimited.
9. Prove request values, successful and nil responses, and collaborator errors
pass through unchanged after permit acquisition.
### Focused Validation
Run:
```sh
gofmt -w internal/capacity/manager.go \
internal/capacity/manager_test.go \
internal/capacity/client.go \
internal/capacity/client_test.go
go test ./internal/capacity
go test -race ./internal/capacity
go vet ./internal/capacity
git diff --check
```
If tests are combined into one file, omit the nonexistent file from `gofmt`.
### Completion Gate
This stage is complete when the standalone component enforces relational
admission and active limits, FIFO cancellation is race-safe, separate pools
are independent, and the wrapper is transparent apart from waiting.
## Stage 3 — Shared Preparation And Early Run Admission
**Status:** Complete.
### Goal
Refactor runner preparation into one shared two-phase pipeline and place
bounded admission after backend resolution but before schema, artifact, and
rendering work.
### Work
1. Add the private pre-admission preparation state and split the existing
`Prepare` logic according to the fixed design.
2. Make `Runner.Prepare` call both phases without an admitter.
3. Add the `RunAdmitter` interface and runner field.
4. Update `NewRunner` and `NewRunnerWithRepairer` to accept the optional
admitter; update internal call sites with nil until root assembly is wired.
5. Change `Runner.Run` to use the first phase, admit by effective backend ID,
defer the returned release, and then use the second phase.
6. Preserve all existing error precedence, target resolution, hashes,
metadata, session behavior, and timing.
7. Return capacity and context errors with the fixed identities. Do not invoke
later collaborators after rejection.
### Tests
1. Keep the existing `Run`/`Prepare` parity coverage passing to prove the
shared phases do not drift.
2. Add a fake admitter that records backend IDs and release calls.
3. Prove a backend-selected run admits with the selected ID even when the
endpoint is overridden.
4. Prove an endpoint-only run uses the unlimited/blank identity and that
`Prepare` never calls admission.
5. Reject admission and assert schema, artifact, renderer, validator, repairer,
and LLM collaborators are not invoked.
6. Prove admission is released after one successful run and representative
second-phase, generation, and validation errors. Prefer a small table around
the single `defer` invariant rather than duplicating every error test.
7. Retain direct-session, backend precedence, credential, hashing, and repair
tests unchanged except for constructor arguments.
### Focused Validation
Run:
```sh
gofmt -w internal/usecase/runner.go \
internal/usecase/runner_test.go
go test ./internal/usecase
go test -race ./internal/usecase
go vet ./internal/usecase
git diff --check
```
### Completion Gate
This stage is complete when `Prepare` remains unrestricted, `Run` admits after
one canonical routing phase and before expensive completion work, every exit
releases admission, and existing preparation semantics remain unchanged.
## Stage 4 — Engine Assembly And Public Runtime Contract
**Status:** Complete.
### Goal
Wire one manager into each engine, schedule built-in and injected clients,
expose the capacity error, and prove assembled runtime behavior.
### Work
1. Add public `ErrCapacityExceeded` and its exact GoDoc in `engine.go`.
2. Map internal capacity exhaustion in `errors.go`.
3. Construct the manager from the registry policy snapshot in `NewEngine`.
4. Wrap the selected internal client after built-in or injected-client
selection and pass the manager and wrapped client to the runner.
5. Update `Engine`, `NewEngine`, `Run`, `WithLLMClient`, and `LLMClient` GoDoc
only where concurrency, capacity, or cancellation statements change.
6. Ensure manager-construction errors match `ErrInvalidConfig`.
7. For internal repair coverage, construct the default repairer with the same
wrapped client used by its runner and confirm repair target backend identity
remains intact.
### Tests
1. Add an external-package assembled test with a small custom limit and a
blocking injected client; assert peak generation equals or remains below
the configured limit.
2. With queue capacity zero, block one accepted run before generation and
assert the next matching-backend run returns `ErrCapacityExceeded`, does not
match `ErrInvalidRequest` or `ErrLLMGenerate`, returns no result, and never
reaches expensive collaborators or the client.
3. In the same or another focused workflow, prove an endpoint override remains
in the selected backend's pool.
4. Prove two engines with the same backend ID have independent capacity.
5. Prove an unlimited custom backend and an endpoint-only profile preserve
concurrent behavior.
6. Cancel a call waiting for an active permit; assert it matches both
`context.Canceled` and `ErrLLMGenerate`, never invokes the injected client,
and leaves capacity reusable.
7. Add one internal repair workflow with concurrent runs or controlled permits
showing initial and repair generations never exceed the same backend limit
and repairs do not perform a second admission.
8. Extend the public error sentinel contract test with
`ErrCapacityExceeded`.
Avoid a second HTTP-level concurrency suite: the capacity client tests and one
assembled injected-client workflow already protect the shared wrapper used by
the built-in client.
### Focused Validation
Run:
```sh
gofmt -w engine.go errors.go backends.go \
internal/usecase/runner.go internal/usecase/runner_test.go \
engine_test.go public_contract_test.go
go test . ./internal/backend ./internal/capacity ./internal/usecase
go test -race . ./internal/capacity ./internal/usecase
go vet . ./internal/backend ./internal/capacity ./internal/usecase
git diff --check
```
Add any newly created capacity files to `gofmt` when they changed in this
stage.
### Completion Gate
This stage is complete when every engine has independent pools, limited runs
are bounded and FIFO at generation, endpoint routing is correct, capacity and
context errors are stable, repairs reuse admission, and both client kinds pass
through the same wrapper.
## Stage 5 — Durable Documentation And Final Validation
**Status:** Complete.
### Goal
Move implemented contracts into their durable owners, record compatibility
impact, and validate the complete repository.
### Work
1. Review every changed exported declaration. Ensure GoDoc is the canonical
owner of exact field types, nil/zero semantics, defaulting, error identity,
engine scope, concurrency safety, cancellation, and source compatibility.
2. Update `doc.go` so its concurrency summary acknowledges backend scheduling
while continuing to require injected collaborators to be concurrency-safe.
3. Update `docs/consumers/pkg-promptkit.md` with task-oriented examples for:
- a limited local backend;
- omitted queue capacity selecting 1024;
- explicit zero queue capacity; and
- handling `ErrCapacityExceeded`.
Keep exact field semantics in GoDoc rather than duplicating a full table.
4. Add `docs/internal/capacity.md` as the durable owner of pool lifecycle,
admission, FIFO active permits, cancellation, client wrapping, and test
ownership.
5. Add `internal/capacity` to `docs/internal/overview.md`.
6. Update `docs/policy/architecture.md` to include the implemented component
and root assembly dependency without turning policy into an API reference.
7. Update `docs/internal/runner.md` to describe the shared two-phase
preparation pipeline, early bounded admission, lease lifetime, generation
permits, repairs, capacity failures, and cancellation.
8. Review `docs/formats.md`; add only a concise link or clarification if needed
to explain that endpoint overrides preserve backend capacity identity.
Do not add concurrency fields to YAML.
9. Do not change the OpenAI-compatible integration contract or
`docs/internal/llm.md` unless implementation changes their current
statements; scheduling is outside the provider wire contract and concrete
model-client implementation.
10. Record in the implementation handoff that built-in OpenRouter now limits
active generations to 16 with queue capacity 1024 and that the release
must be a pre-`v1` minor release. Do not edit the release procedure or
create a tag.
11. After every check passes, set `concurrency.md`, this implementation plan,
and each stage status to `Complete`. Do not remove the roadmaps in the
implementation change; lifecycle retirement follows review.
### Full Validation
Run the complete sequence from `docs/development.md`:
```sh
go test ./...
go test -race ./...
go vet ./...
go build ./...
go run ./examples/go-library/prepare
gofmt -l $(git ls-files '*.go')
git diff --check
```
The formatting command must produce no paths. Follow every added or changed
Markdown link and confirm its target and heading exist.
Also inspect:
```sh
git status --short
git diff --stat
git diff
```
Confirm that:
- only intended backend, capacity, runner, facade, test, documentation, and
roadmap files changed;
- no `go.work`, `go.work.sum`, local module replacement, credential,
generated binary, coverage output, or unrelated change was introduced;
- the built-in OpenRouter policy is exactly 16/1024;
- custom and endpoint-only backends remain unlimited by omission;
- explicit queue zero is distinguishable from omission;
- no capacity value enters execution targets, generated requests, stable JSON,
prompt/profile YAML, or provider payloads;
- every engine owns distinct pools with no package-global mutable state;
- every initial and repair generation uses the active permit wrapper;
- capacity and waiter state is released on success, error, panic unwinding,
and cancellation;
- concurrency tests use deterministic coordination rather than sleeps;
- current-state documentation describes implemented behavior rather than
referring readers to the roadmaps; and
- the feature and implementation roadmaps contain no unresolved work marked
complete.
### Completion Gate
The implementation is complete only when every target-end-state item in
`concurrency.md` is implemented, race-enabled tests demonstrate the configured
limits and cancellation safety, durable contracts no longer depend on roadmap
prose, and the OpenRouter compatibility change is clearly reported for the
next minor release.
## Implementation Handoff
Backend-specific capacity management is implemented and has passed the complete
repository validation sequence. The built-in OpenRouter backend now permits 16
active generations and a waiting capacity of 1024. Custom backends remain
unlimited when their limit is omitted, and endpoint-only profiles remain
unlimited.
Publishing this behavior requires a pre-`v1` minor release. Its release notes
must identify that unusually high concurrent OpenRouter use can now wait or
return `ErrCapacityExceeded`. This implementation does not change a module
version or create a tag.
## Open Questions
None. The feature roadmap and this plan fix the public representation,
registry defaults, admission bound, FIFO generation behavior, early-routing
refactor, cancellation races, error identities, engine and repair lifetimes,
test ownership, compatibility treatment, and non-goals required for
implementation.

406
engine.go
View File

@@ -11,14 +11,16 @@ import (
"strings" "strings"
"time" "time"
openrouter "gitea.maximumdirect.net/eric/promptkit-backend-openrouter"
rakestrawhome "gitea.maximumdirect.net/eric/promptkit-backend-rakestrawhome"
artifactadapter "gitea.maximumdirect.net/eric/promptkit/internal/artifact" artifactadapter "gitea.maximumdirect.net/eric/promptkit/internal/artifact"
"gitea.maximumdirect.net/eric/promptkit/internal/backend" "gitea.maximumdirect.net/eric/promptkit/internal/backend"
"gitea.maximumdirect.net/eric/promptkit/internal/capacity" "gitea.maximumdirect.net/eric/promptkit/internal/capacity"
"gitea.maximumdirect.net/eric/promptkit/internal/catalog"
"gitea.maximumdirect.net/eric/promptkit/internal/defaults" "gitea.maximumdirect.net/eric/promptkit/internal/defaults"
"gitea.maximumdirect.net/eric/promptkit/internal/domain" "gitea.maximumdirect.net/eric/promptkit/internal/domain"
"gitea.maximumdirect.net/eric/promptkit/internal/llm" "gitea.maximumdirect.net/eric/promptkit/internal/llm"
"gitea.maximumdirect.net/eric/promptkit/internal/profile" "gitea.maximumdirect.net/eric/promptkit/internal/profile"
"gitea.maximumdirect.net/eric/promptkit/internal/profile/builtin"
"gitea.maximumdirect.net/eric/promptkit/internal/prompt" "gitea.maximumdirect.net/eric/promptkit/internal/prompt"
"gitea.maximumdirect.net/eric/promptkit/internal/promptdef" "gitea.maximumdirect.net/eric/promptkit/internal/promptdef"
"gitea.maximumdirect.net/eric/promptkit/internal/usecase" "gitea.maximumdirect.net/eric/promptkit/internal/usecase"
@@ -53,9 +55,9 @@ var (
// an execution profile or resolve its backend, except for the profile // an execution profile or resolve its backend, except for the profile
// not-found case represented by ErrProfileNotFound. // not-found case represented by ErrProfileNotFound.
ErrProfileLoad = errors.New("failed to load execution profile") ErrProfileLoad = errors.New("failed to load execution profile")
// ErrAPIKeyEnvMissing identifies an APIKeyEnv whose environment variable is // ErrAPIKeyEnvMissing identifies an explicitly required APIKeyEnv whose
// unset or empty when no direct RunRequest.APIKey takes precedence. Such an // environment variable is unset or empty after direct RunRequest.APIKey
// error also matches ErrInvalidRequest. // precedence is applied. Such an error also matches ErrInvalidRequest.
ErrAPIKeyEnvMissing = errors.New("api_key_env points to an unset environment variable") ErrAPIKeyEnvMissing = errors.New("api_key_env points to an unset environment variable")
// ErrArtifactLoad identifies a failure to resolve an input artifact. Errors // ErrArtifactLoad identifies a failure to resolve an input artifact. Errors
// returned by an injected ArtifactReader remain available through errors.Is. // returned by an injected ArtifactReader remain available through errors.Is.
@@ -63,13 +65,15 @@ var (
// ErrPromptRender identifies a failure to render prompt messages or the // ErrPromptRender identifies a failure to render prompt messages or the
// session ID from the resolved inputs and variables. // session ID from the resolved inputs and variables.
ErrPromptRender = errors.New("failed to render prompt") ErrPromptRender = errors.New("failed to render prompt")
// ErrCapacityExceeded identifies a Run rejected because the selected backend // ErrCapacityExceeded identifies a Run or RunPrepared rejected because the
// already admitted ConcurrencyLimit + QueueCapacity calls. It is not an // selected backend already admitted ConcurrencyLimit + QueueCapacity calls.
// invalid request, an LLM or provider rate-limit response, or ErrLLMGenerate. // A [CapacityError] reports the selected backend ID. It is not an invalid
// request, an LLM or provider rate-limit response, or ErrLLMGenerate.
ErrCapacityExceeded = errors.New("backend capacity exceeded") ErrCapacityExceeded = errors.New("backend capacity exceeded")
// ErrLLMGenerate identifies a model-client failure or a nil successful // ErrLLMGenerate identifies a model-client failure or a nil successful
// response. Errors returned by an injected LLMClient remain available // response. A built-in OpenAI-compatible non-2xx response is available as a
// through errors.Is. // [GenerationError]. Errors returned by an injected LLMClient remain
// available through errors.Is.
ErrLLMGenerate = errors.New("failed to generate output") ErrLLMGenerate = errors.New("failed to generate output")
// ErrValidation identifies an operational failure to load or compile a // ErrValidation identifies an operational failure to load or compile a
// schema or validate output. A completed validation whose Status is // schema or validate output. A completed validation whose Status is
@@ -77,12 +81,15 @@ var (
ErrValidation = errors.New("failed to validate output") ErrValidation = errors.New("failed to validate output")
) )
// Engine prepares and runs Promptkit prompt requests. // Engine inspects prompts and profiles and prepares and runs Promptkit prompt
// requests.
// //
// An Engine is safe for concurrent calls to [Engine.Prepare] and [Engine.Run]. // An Engine is safe for concurrent calls to [Engine.InspectPrompt],
// Each Engine owns independent backend-capacity pools that coordinate Run // [Engine.InspectProfile], [Engine.Prepare], [Engine.PrepareExecution],
// admission and model generation. Injected collaborators may still be invoked // [Engine.Run], and [Engine.RunPrepared]. Each Engine owns independent
// concurrently across different backend pools or for unlimited backends. // backend-capacity pools that coordinate Run and RunPrepared admission and
// model generation. Injected collaborators may still be invoked concurrently
// across different backend pools or for unlimited backends.
type Engine struct { type Engine struct {
runner *usecase.Runner runner *usecase.Runner
} }
@@ -94,9 +101,10 @@ type Config struct {
// It is required unless a WithPromptFS or WithPromptFile option supplies the // It is required unless a WithPromptFS or WithPromptFile option supplies the
// prompt source. // prompt source.
PromptDir string PromptDir string
// ProfileDir is an optional directory whose profiles take precedence over // ProfileDir is an optional ordinary configured source whose profiles take
// embedded built-in profiles. An empty value selects only built-ins unless // precedence over application fallback and maintained catalog profiles. An
// profile options are also supplied. // empty value selects the lower-precedence sources unless a profile-source
// option supplies the ordinary source.
ProfileDir string ProfileDir string
// SchemaDir is the root for JSON Schema files. An empty value uses the // SchemaDir is the root for JSON Schema files. An empty value uses the
// current directory. WithSchemaFS or WithSchemaFile replaces this source. // current directory. WithSchemaFS or WithSchemaFile replaces this source.
@@ -115,12 +123,12 @@ type Config struct {
// Option customizes engine construction. // Option customizes engine construction.
// //
// NewEngine applies options in argument order and ignores nil options. Within // NewEngine applies options in argument order and ignores nil options. Within
// each prompt-source, profile-source, in-memory-profile, schema-source, // each prompt-source, ordinary-profile-source, fallback-profile-source,
// model-client, and artifact-reader category, the last non-nil valid option // in-memory-profile, schema-source, model-client, and artifact-reader
// replaces earlier options in that category. WithBackend is the additive // category, the last non-nil valid option replaces earlier options in that
// exception: unique registrations accumulate, and a repeated backend ID is an // category. WithBackend is the additive exception: unique registrations
// error rather than a replacement. An invalid option fails construction even // accumulate, and a repeated backend ID is an error rather than a replacement.
// if a later option would replace it. // An invalid option fails construction even if a later option would replace it.
type Option interface { type Option interface {
apply(*engineOptions) error apply(*engineOptions) error
} }
@@ -132,26 +140,30 @@ func (f optionFunc) apply(options *engineOptions) error {
} }
type engineOptions struct { type engineOptions struct {
llmClient llm.Client llmClient llm.Client
artifactReader artifactadapter.Reader artifactReader artifactadapter.Reader
promptDefs promptdef.Repository promptDefs promptdef.Repository
profiles profile.Repository profiles profile.Repository
memoryProfiles profile.Repository fallbackProfiles profile.Repository
backends []domain.Backend memoryProfiles profile.Repository
validator validate.Validator backends []domain.Backend
promptSource bool validator validate.Validator
profileSource bool promptSource bool
memorySource bool profileSource bool
validatorSource bool fallbackProfileSource bool
artifactSource bool memorySource bool
validatorSource bool
artifactSource bool
} }
// WithLLMClient replaces the built-in model client used by [Engine.Run]. // WithLLMClient replaces the built-in model client used by [Engine.Run] and
// [Engine.RunPrepared].
// //
// A nil client makes NewEngine fail with ErrInvalidConfig. The Engine schedules // A nil client makes NewEngine fail with ErrInvalidConfig. The Engine schedules
// Generate calls according to the selected backend's capacity policy, but the // Generate calls according to the selected backend's capacity policy, but the
// client may still be called concurrently across different backend pools or for // client may still be called concurrently across different backend pools or for
// unlimited backends. The client is not used by [Engine.Prepare]. // unlimited backends. The client is not used by [Engine.Prepare] or
// [Engine.PrepareExecution].
func WithLLMClient(client LLMClient) Option { func WithLLMClient(client LLMClient) Option {
return optionFunc(func(options *engineOptions) error { return optionFunc(func(options *engineOptions) error {
if client == nil { if client == nil {
@@ -210,7 +222,7 @@ func WithPromptFile(path string) Option {
if err != nil { if err != nil {
return err return err
} }
options.promptDefs = promptdef.NewFSRepository(fsys, root) options.promptDefs = promptdef.NewFileRepository(fsys, root, filepath.Dir(path))
options.promptSource = true options.promptSource = true
return nil return nil
}) })
@@ -218,12 +230,12 @@ func WithPromptFile(path string) Option {
// WithProfileFS loads execution profiles from fsys under root. // WithProfileFS loads execution profiles from fsys under root.
// //
// Profiles from this source overlay built-in profiles. Profile YAML must use // Profiles from this ordinary configured source take precedence over
// api_key_env for environment-based credentials; raw API keys are rejected. // application fallback and built-in profiles. Profile YAML must use api_key_env
// fsys must be non-nil and root must be non-empty; otherwise NewEngine fails // for environment-based credentials; raw API keys are rejected. fsys must be
// with ErrInvalidConfig. This option replaces Config.ProfileDir and earlier // non-nil and root must be non-empty; otherwise NewEngine fails with
// file or FS profile-source options, but remains below WithProfiles in // ErrInvalidConfig. This option replaces Config.ProfileDir and earlier file or
// precedence. // FS profile-source options, but remains below WithProfiles in precedence.
func WithProfileFS(fsys fs.FS, root string) Option { func WithProfileFS(fsys fs.FS, root string) Option {
return optionFunc(func(options *engineOptions) error { return optionFunc(func(options *engineOptions) error {
if fsys == nil { if fsys == nil {
@@ -240,11 +252,12 @@ func WithProfileFS(fsys fs.FS, root string) Option {
// WithProfileFile loads execution profiles from the single profile file at path. // WithProfileFile loads execution profiles from the single profile file at path.
// //
// The profile overlays built-in profiles. Profile YAML must use api_key_env for // The profile takes precedence over application fallback and built-in profiles.
// environment-based credentials; raw API keys are rejected. path must name an // Profile YAML must use api_key_env for environment-based credentials; raw API
// existing non-directory file when NewEngine applies the option. This option // keys are rejected. path must name an existing non-directory file when
// replaces Config.ProfileDir and earlier file or FS profile-source options, // NewEngine applies the option. This option replaces Config.ProfileDir and
// but remains below WithProfiles in precedence. // earlier file or FS profile-source options, but remains below WithProfiles in
// precedence.
func WithProfileFile(path string) Option { func WithProfileFile(path string) Option {
return optionFunc(func(options *engineOptions) error { return optionFunc(func(options *engineOptions) error {
fsys, root, err := fileSource(path) fsys, root, err := fileSource(path)
@@ -257,13 +270,48 @@ func WithProfileFile(path string) Option {
}) })
} }
// WithProfiles configures in-memory profiles that take precedence over // WithFallbackProfileFS supplies application-owned fallback profile
// configured profile files and built-in profiles. // definitions from fsys under root.
// //
// NewEngine validates and copies every profile. IDs must be unique within one // Profile lookup checks, in order, profiles supplied by WithProfiles; the
// call. An invalid profile, duplicate ID, or unsupported ExtraParams value // ordinary configured source selected by WithProfileFile, WithProfileFS, or
// makes construction fail with ErrInvalidConfig. Repeating WithProfiles // Config.ProfileDir; this fallback source; and Promptkit's maintained catalog
// replaces the complete earlier in-memory set rather than merging it. // profiles. Each source supplies a complete profile definition; profile fields
// are not merged between sources. Only an absent profile ID proceeds to the
// next source. A matching read, parse, duplicate, validation, or credential
// format failure stops resolution.
//
// Files use the ordinary strict profile YAML and api_key_env credential rules.
// Loading and validation are lazy: NewEngine validates this option's arguments
// but does not read profile files. fsys must be non-nil and root must be
// nonblank; otherwise NewEngine returns an error matching ErrInvalidConfig.
// Repeating this option replaces the earlier valid fallback source.
//
// This option controls profile-definition lookup, not provider or generation
// failover.
func WithFallbackProfileFS(fsys fs.FS, root string) Option {
return optionFunc(func(options *engineOptions) error {
if fsys == nil {
return ErrInvalidConfig
}
if strings.TrimSpace(root) == "" {
return ErrInvalidConfig
}
options.fallbackProfiles = profile.NewFSRepository(fsys, root)
options.fallbackProfileSource = true
return nil
})
}
// WithProfiles configures in-memory profiles that take precedence over
// ordinary configured, application fallback, and built-in profiles.
//
// NewEngine locally validates and copies every profile. IDs must be unique
// within one call. An invalid local definition, duplicate ID, or unsupported
// ExtraParams value makes construction fail with ErrInvalidConfig. A derived
// profile's base reference and resolved target completeness are checked when it
// is selected or inspected. Repeating WithProfiles replaces the complete
// earlier in-memory set rather than merging it.
func WithProfiles(profiles ...Profile) Option { func WithProfiles(profiles ...Profile) Option {
return optionFunc(func(options *engineOptions) error { return optionFunc(func(options *engineOptions) error {
repo, err := newMemoryProfileRepository(profiles) repo, err := newMemoryProfileRepository(profiles)
@@ -344,15 +392,17 @@ func NewEngine(cfg Config, opts ...Option) (*Engine, error) {
promptDefs = promptdef.NewFilesystemRepository(cfg.PromptDir) promptDefs = promptdef.NewFilesystemRepository(cfg.PromptDir)
} }
profiles := builtin.NewRepositoryWithDirectory(cfg.ProfileDir) maintainedCatalogs, err := catalog.Load(
if options.profileSource { catalog.Source{Name: "OpenRouter", ExpectedBackendID: backend.OpenRouterID, FS: openrouter.FS(), Root: openrouter.Root},
profiles = builtin.NewRepositoryWithPrimary(options.profiles) catalog.Source{Name: "Rakestrawhome", ExpectedBackendID: backend.RakestrawHomeID, FS: rakestrawhome.FS(), Root: rakestrawhome.Root},
} )
if options.memorySource { if err != nil {
profiles = profile.NewOverlayRepository(options.memoryProfiles, profiles) return nil, fmt.Errorf("%w: failed to load maintained catalogs: %v", ErrInvalidConfig, err)
} }
backendRegistry, err := backend.NewRegistry(options.backends) profiles := newProfileRepository(cfg.ProfileDir, options, maintainedCatalogs.Profiles)
backendRegistry, err := backend.NewRegistry(maintainedCatalogs.Backends, options.backends)
if err != nil { if err != nil {
return nil, fmt.Errorf("%w: failed to construct backend registry: %v", ErrInvalidConfig, err) return nil, fmt.Errorf("%w: failed to construct backend registry: %v", ErrInvalidConfig, err)
} }
@@ -390,7 +440,7 @@ func NewEngine(cfg Config, opts ...Option) (*Engine, error) {
} }
return &Engine{ return &Engine{
runner: usecase.NewRunner( runner: usecase.NewRunnerWithRepairer(
promptDefs, promptDefs,
profiles, profiles,
backendRegistry, backendRegistry,
@@ -398,32 +448,141 @@ func NewEngine(cfg Config, opts ...Option) (*Engine, error) {
prompt.NewGoRenderer(), prompt.NewGoRenderer(),
llmClient, llmClient,
validator, validator,
usecase.NewDefaultOutputRepairer(llmClient),
capacityManager, capacityManager,
), ),
}, nil }, nil
} }
func newProfileRepository(profileDir string, options engineOptions, maintained profile.Repository) profile.Repository {
repository := maintained
if options.fallbackProfileSource {
repository = profile.NewOverlayRepository(options.fallbackProfiles, repository)
}
if options.profileSource {
repository = profile.NewOverlayRepository(options.profiles, repository)
} else if strings.TrimSpace(profileDir) != "" {
repository = profile.NewOverlayRepository(profile.NewFilesystemRepository(profileDir), repository)
}
if options.memorySource {
repository = profile.NewOverlayRepository(options.memoryProfiles, repository)
}
return profile.NewResolvingRepository(repository)
}
func fileSource(name string) (fs.FS, string, error) { func fileSource(name string) (fs.FS, string, error) {
cleanName := strings.TrimSpace(name) if strings.TrimSpace(name) == "" {
if cleanName == "" {
return nil, "", ErrInvalidConfig return nil, "", ErrInvalidConfig
} }
dir := filepath.Dir(cleanName) dir := filepath.Dir(name)
base := filepath.Base(cleanName) base := filepath.Base(name)
if base == "." || base == string(filepath.Separator) || strings.TrimSpace(base) == "" { if base == "." || base == string(filepath.Separator) {
return nil, "", ErrInvalidConfig return nil, "", ErrInvalidConfig
} }
info, err := os.Stat(cleanName) info, err := os.Stat(name)
if err != nil { if err != nil {
return nil, "", fmt.Errorf("%w: failed to access source file %q: %v", ErrInvalidConfig, cleanName, err) return nil, "", fmt.Errorf("%w: failed to access source file %q: %v", ErrInvalidConfig, name, err)
} }
if info.IsDir() { if info.IsDir() {
return nil, "", fmt.Errorf("%w: source path %q must be a file", ErrInvalidConfig, cleanName) return nil, "", fmt.Errorf("%w: source path %q must be a file", ErrInvalidConfig, name)
} }
return os.DirFS(dir), filepath.ToSlash(base), nil return os.DirFS(dir), filepath.ToSlash(base), nil
} }
// InspectPrompt resolves one explicit prompt definition without selecting a
// profile or starting execution work.
//
// InspectPrompt requires a nonblank promptID. It passes nonblank promptID and
// promptVersion values unchanged to the engine's ordinary, case-sensitive
// prompt selection. An empty version succeeds only when that source has one
// selected ID; a nonempty version selects one exact ID/version pair. The
// configured prompt source is used without merging, fallback, or enumeration.
//
// A successful result proves that the selected definition and any referenced
// message content files were structurally loaded. Inputs are returned in
// definition order. DefaultProfileID is declared metadata only and is not
// resolved. OutputContract is the normalized declared contract, with a JSON
// Schema path when declared but without loading or compiling that schema.
// PromptHash is the same opaque equality value as PreparedRun.PromptHash for
// the selected definition and observed source state; its spelling, length,
// encoding, algorithm, and security properties are not contracts.
//
// This method does not return prompt bodies, templates, source paths, schemas,
// rendered messages, or execution settings. It does not resolve a profile or
// credential, read artifacts or schemas, render, validate, admit capacity,
// contact a provider, or generate model output. The returned PromptInspection
// and its input slice are caller-owned. Filesystem-backed inspection is a
// point-in-time lookup and does not freeze a definition for later execution.
//
// A nil Engine returns an error matching ErrInvalidConfig. A blank prompt ID
// matches ErrInvalidRequest. An absent exact ID or version matches
// ErrPromptNotFound and not ErrPromptLoad. Malformed, unreadable, duplicate,
// ambiguous, referenced-content, or hashing failures match ErrPromptLoad.
// Cancellation during lookup matches ErrPromptLoad while preserving the
// context error. InspectPrompt returns no partial result on error.
func (e *Engine) InspectPrompt(
ctx context.Context,
promptID string,
promptVersion string,
) (*PromptInspection, error) {
if e == nil || e.runner == nil {
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
}
inspection, err := e.runner.InspectPrompt(ctx, promptID, promptVersion)
if err != nil {
return nil, mapPublicError(err)
}
return fromDomainPromptInspection(inspection), nil
}
// InspectProfile resolves one explicit profile without selecting a prompt or
// starting execution work.
//
// InspectProfile trims surrounding whitespace from profileID and looks up the
// resulting nonblank ID exactly and case-sensitively through the engine's
// in-memory, ordinary configured-source, application fallback, and built-in
// profile precedence. It applies the framework timeout baseline, selected
// backend, and then selected profile to EffectiveModelParams without a request
// override. BackendID is empty for an endpoint-only profile.
//
// APIKeyEnv in the returned target is an environment-variable name, never its
// value. APIKeyRequired instead reports a direct credential requirement and is
// mutually exclusive with a nonblank APIKeyEnv. InspectProfile neither derives
// an ID from a prompt default_profile nor checks credential availability, so an
// absent or blank named environment variable is not an error.
//
// The returned ProfileInspection and all nested mutable values are
// caller-owned. Filesystem-backed inspection is a point-in-time lookup and
// does not freeze the profile for a later execution. This method does not load
// a prompt, render, read artifacts or schemas, admit backend capacity, contact
// a provider, or generate model output.
//
// A nil Engine returns an error matching ErrInvalidConfig. A blank profile ID
// matches ErrInvalidRequest. An absent exact ID matches ErrProfileNotFound and
// not ErrProfileLoad. Malformed or unreadable profile data, an unknown backend,
// or an invalid resolved target matches ErrProfileLoad. Cancellation during
// profile loading matches ErrProfileLoad while preserving the context error.
// InspectProfile returns no partial result on error.
func (e *Engine) InspectProfile(ctx context.Context, profileID string) (*ProfileInspection, error) {
if e == nil || e.runner == nil {
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
}
inspection, err := e.runner.InspectProfile(ctx, profileID)
if err != nil {
return nil, mapPublicError(err)
}
return fromDomainProfileInspection(inspection), nil
}
// Prepare resolves and renders a prompt request without calling an LLM. // Prepare resolves and renders a prompt request without calling an LLM.
// It appends any validated RunRequest.AppendedMessages after rendered
// definition messages in the returned caller-owned snapshot.
// //
// Prepare selects the prompt and profile, resolves any selected backend and // Prepare selects the prompt and profile, resolves any selected backend and
// effective execution settings, resolves the output contract, loads and hashes // effective execution settings, resolves the output contract, loads and hashes
@@ -457,24 +616,63 @@ func (e *Engine) Prepare(ctx context.Context, req RunRequest) (*PreparedRun, err
return fromDomainPreparedRun(prepared), nil return fromDomainPreparedRun(prepared), nil
} }
// PrepareExecution completely prepares a prompt request without calling the
// configured LLMClient or reserving backend admission capacity. Validated
// RunRequest.AppendedMessages are included in the frozen effective messages.
//
// The returned opaque handle is bound to this Engine and permits one
// [Engine.RunPrepared] invocation. Preparation freezes the selected sources,
// rendered messages, effective settings, inputs, provider structured-output
// metadata, and validation resources needed by that invocation. The handle
// retains a direct RunRequest.APIKey only in private execution state;
// [PreparedExecution.Details] is credential-redacted.
//
// The context governs preparation only. Cancellation after this method
// returns does not invalidate the handle or propagate to RunPrepared.
// PrepareExecution returns the same error categories as [Engine.Prepare] and
// returns no handle on error. A nil Engine returns an error matching
// ErrInvalidConfig.
func (e *Engine) PrepareExecution(ctx context.Context, req RunRequest) (*PreparedExecution, error) {
if e == nil || e.runner == nil {
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
}
domainReq, err := toDomainRunRequest(req)
if err != nil {
return nil, fmt.Errorf("%w: %v", ErrInvalidRequest, err)
}
prepared, err := e.runner.PrepareExecution(ctx, domainReq)
if err != nil {
return nil, mapPublicError(err)
}
return &PreparedExecution{internal: prepared}, nil
}
// Run prepares a request, invokes the configured LLMClient, and validates the // Run prepares a request, invokes the configured LLMClient, and validates the
// generated output. // generated output. Each call resolves current sources and composes a fresh,
// stateless effective prompt with any validated RunRequest.AppendedMessages.
// //
// A content-validation failure is a successful run whose // A content-validation failure is a successful run whose
// RunResult.Validation has Status ValidationFailed. An inability to perform // RunResult.Validation has Status ValidationFailed. When its output contract
// validation returns an error matching ErrValidation and no partial result. // has a positive repair budget, a failed eligible validation can make bounded
// The public Engine does not perform output repair, so validation is // additional model calls and stops at the first valid candidate. Exhaustion
// single-pass even when OutputContract.RepairAttempts is positive. // returns the final failed validation result with cumulative usage and actual
// repair attempts. An inability to generate or validate returns an error and
// no partial result.
// //
// Run can return every error category documented by [Engine.Prepare], plus // Run can return every error category documented by [Engine.Prepare], plus
// ErrCapacityExceeded and ErrLLMGenerate. ErrCapacityExceeded identifies // ErrCapacityExceeded and ErrLLMGenerate. An engine admission rejection is
// rejection before artifacts, schemas, rendering, or model generation because // discoverable as [CapacityError] and still matches ErrCapacityExceeded. It
// the selected backend's admission capacity is full; it does not match // occurs before artifacts, schemas, rendering, or model generation because the
// ErrInvalidRequest or ErrLLMGenerate. Errors from injected clients remain // selected backend's admission capacity is full; it does not match
// available through errors.Is. Cancellation while waiting for model-generation // ErrInvalidRequest or ErrLLMGenerate. A built-in OpenAI-compatible non-2xx
// capacity matches both ErrLLMGenerate and the context error. Cancellation // response is discoverable as [GenerationError]. Errors from injected clients
// otherwise follows the active collaborator's documented behavior. A nil // remain available through errors.Is. Cancellation while waiting for
// Engine returns ErrInvalidConfig. Run returns no partial result on error. // model-generation capacity matches both ErrLLMGenerate and the context error.
// Cancellation otherwise follows the active collaborator's documented
// behavior. A nil Engine returns ErrInvalidConfig. Run returns no partial
// result on error.
func (e *Engine) Run(ctx context.Context, req RunRequest) (*RunResult, error) { func (e *Engine) Run(ctx context.Context, req RunRequest) (*RunResult, error) {
if e == nil || e.runner == nil { if e == nil || e.runner == nil {
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig) return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
@@ -491,3 +689,43 @@ func (e *Engine) Run(ctx context.Context, req RunRequest) (*RunResult, error) {
} }
return fromDomainRunResult(result), nil return fromDomainRunResult(result), nil
} }
// RunPrepared atomically claims and executes a handle created by
// [Engine.PrepareExecution].
//
// A valid owning-Engine invocation consumes the handle's one attempt before
// credential revalidation, backend admission, generation, or validation.
// Cancellation, capacity rejection, generation failure, operational
// validation failure, and success all leave the handle unusable. A nil,
// zero-value, foreign-Engine, discarded, claimed, or used handle returns an
// error matching ErrInvalidRequest; a nil Engine returns ErrInvalidConfig and
// does not claim the handle.
//
// The supplied context governs this execution attempt independently of the
// preparation context. It covers credential revalidation, admission,
// generation, validation, and any bounded output repair. Result timing begins
// after the claim and excludes preparation and consumer-held delay.
//
// RunPrepared can return ErrInvalidRequest, ErrAPIKeyEnvMissing,
// ErrCapacityExceeded, ErrLLMGenerate, or ErrValidation as applicable while
// preserving documented collaborator and context identities. An engine
// admission rejection is discoverable as [CapacityError] and still matches
// ErrCapacityExceeded. A built-in OpenAI-compatible non-2xx response is
// discoverable as [GenerationError]. A completed content-validation rejection,
// including repair exhaustion, is returned in RunResult, not as an operational
// error. An operational error returns no partial RunResult.
func (e *Engine) RunPrepared(ctx context.Context, prepared *PreparedExecution) (*RunResult, error) {
if e == nil || e.runner == nil {
return nil, fmt.Errorf("%w: engine is nil", ErrInvalidConfig)
}
var internal *usecase.PreparedExecution
if prepared != nil {
internal = prepared.internal
}
result, err := e.runner.RunPrepared(ctx, internal)
if err != nil {
return nil, mapPublicError(err)
}
return fromDomainRunResult(result), nil
}

File diff suppressed because it is too large Load Diff

View File

@@ -3,8 +3,10 @@ package promptkit
import ( import (
"errors" "errors"
"fmt" "fmt"
"strings"
"gitea.maximumdirect.net/eric/promptkit/internal/capacity" "gitea.maximumdirect.net/eric/promptkit/internal/capacity"
"gitea.maximumdirect.net/eric/promptkit/internal/llm"
"gitea.maximumdirect.net/eric/promptkit/internal/profile" "gitea.maximumdirect.net/eric/promptkit/internal/profile"
"gitea.maximumdirect.net/eric/promptkit/internal/promptdef" "gitea.maximumdirect.net/eric/promptkit/internal/promptdef"
"gitea.maximumdirect.net/eric/promptkit/internal/usecase" "gitea.maximumdirect.net/eric/promptkit/internal/usecase"
@@ -14,7 +16,25 @@ func mapPublicError(err error) error {
if err == nil { if err == nil {
return nil return nil
} }
var internalCapacityError *usecase.CapacityError
if errors.As(err, &internalCapacityError) && internalCapacityError != nil &&
strings.TrimSpace(internalCapacityError.BackendID) != "" {
return &CapacityError{BackendID: internalCapacityError.BackendID}
}
publicErr := publicErrorFor(err) publicErr := publicErrorFor(err)
var providerHTTPError *llm.ProviderHTTPError
if errors.As(err, &providerHTTPError) && providerHTTPError != nil {
generationErr := newGenerationError(
providerHTTPError.StatusCode(),
providerHTTPError.ProviderCode(),
providerHTTPError.ProviderType(),
providerHTTPError.ProviderMessage(),
)
if publicErr != nil && !errors.Is(publicErr, ErrLLMGenerate) {
return fmt.Errorf("%w: %w", publicErr, generationErr)
}
return generationErr
}
if publicErr == nil { if publicErr == nil {
return err return err
} }

View File

@@ -6,6 +6,7 @@ import (
"fmt" "fmt"
"testing" "testing"
"gitea.maximumdirect.net/eric/promptkit/internal/llm"
"gitea.maximumdirect.net/eric/promptkit/internal/usecase" "gitea.maximumdirect.net/eric/promptkit/internal/usecase"
) )
@@ -20,3 +21,55 @@ func TestMapPublicErrorPreservesGenerationCancellation(t *testing.T) {
t.Fatalf("mapped error=%v, want context.Canceled", err) t.Fatalf("mapped error=%v, want context.Canceled", err)
} }
} }
func TestMapPublicErrorTranslatesCapacityError(t *testing.T) {
internalErr := &usecase.CapacityError{BackendID: "limited"}
err := mapPublicError(internalErr)
var publicErr *CapacityError
if !errors.As(err, &publicErr) || publicErr == nil {
t.Fatalf("mapped error=%v, want public CapacityError", err)
}
if publicErr.BackendID != "limited" {
t.Fatalf("mapped backend ID=%q, want limited", publicErr.BackendID)
}
if !errors.Is(err, ErrCapacityExceeded) {
t.Fatalf("mapped error=%v, want ErrCapacityExceeded", err)
}
if errors.Is(err, ErrInvalidRequest) || errors.Is(err, ErrLLMGenerate) {
t.Fatalf("mapped capacity error has an unrelated category: %v", err)
}
var leakedInternalErr *usecase.CapacityError
if errors.As(err, &leakedInternalErr) {
t.Fatalf("mapped error exposes internal CapacityError: %v", err)
}
internalErr.BackendID = "changed"
if publicErr.BackendID != "limited" {
t.Fatalf("mapped backend ID changed with source error: %q", publicErr.BackendID)
}
}
func TestMapPublicErrorPreservesValidationAroundGenerationError(t *testing.T) {
internalErr := fmt.Errorf(
"%w: %w",
usecase.ErrValidation,
&llm.ProviderHTTPError{},
)
err := mapPublicError(internalErr)
if !errors.Is(err, ErrValidation) {
t.Fatalf("mapped error=%v, want ErrValidation", err)
}
if !errors.Is(err, ErrLLMGenerate) {
t.Fatalf("mapped error=%v, want ErrLLMGenerate", err)
}
var generationErr *GenerationError
if !errors.As(err, &generationErr) || generationErr == nil {
t.Fatalf("mapped error=%v, want GenerationError", err)
}
var leakedInternalErr *llm.ProviderHTTPError
if errors.As(err, &leakedInternalErr) {
t.Fatalf("mapped error exposes internal ProviderHTTPError: %v", err)
}
}

View File

@@ -18,7 +18,7 @@ func (r RunRequest) GoString() string {
func (r RunRequest) redactedString() string { func (r RunRequest) redactedString() string {
return fmt.Sprintf( return fmt.Sprintf(
"promptkit.RunRequest{PromptID:%q PromptVersion:%q ProfileID:%q APIKeySet:%t Inputs:%d Vars:%d ExecutionSet:%t ValidationSet:%t}", "promptkit.RunRequest{PromptID:%q PromptVersion:%q ProfileID:%q APIKeySet:%t Inputs:%d Vars:%d ExecutionSet:%t ValidationSet:%t AppendedMessages:%d}",
r.PromptID, r.PromptID,
r.PromptVersion, r.PromptVersion,
r.ProfileID, r.ProfileID,
@@ -27,6 +27,7 @@ func (r RunRequest) redactedString() string {
len(r.Vars), len(r.Vars),
r.Execution != nil, r.Execution != nil,
r.Validation != nil, r.Validation != nil,
len(r.AppendedMessages),
) )
} }

87
generation_error.go Normal file
View File

@@ -0,0 +1,87 @@
package promptkit
import "fmt"
// GenerationError reports a non-2xx response from Promptkit's built-in
// OpenAI-compatible client during [Engine.Run] or [Engine.RunPrepared].
//
// Engine-produced values are immutable, caller-owned values. Use errors.Is to
// match [ErrLLMGenerate] and errors.As with a *GenerationError target to obtain
// this type. The four provider accessors expose untrusted provider-controlled
// values that can contain sensitive request or schema fragments. Applications
// must apply their own disclosure policy before logging, displaying, or
// returning them to another caller.
//
// Accessors, Error, GoString, and Unwrap are safe on a nil receiver and a zero
// value. Default and Go-syntax formatting deliberately redact provider details.
// GenerationError has no stable JSON representation.
type GenerationError struct {
statusCode int
providerCode string
providerType string
providerMessage string
}
func newGenerationError(statusCode int, providerCode, providerType, providerMessage string) *GenerationError {
return &GenerationError{
statusCode: statusCode,
providerCode: providerCode,
providerType: providerType,
providerMessage: providerMessage,
}
}
// StatusCode returns the received provider HTTP status code, or zero for a nil
// receiver or zero value.
func (e *GenerationError) StatusCode() int {
if e == nil {
return 0
}
return e.statusCode
}
// ProviderCode returns the normalized provider error code, if present. Its
// value is untrusted and may contain sensitive data.
func (e *GenerationError) ProviderCode() string {
if e == nil {
return ""
}
return e.providerCode
}
// ProviderType returns the normalized provider error type, if present. Its
// value is untrusted and may contain sensitive data.
func (e *GenerationError) ProviderType() string {
if e == nil {
return ""
}
return e.providerType
}
// ProviderMessage returns the bounded normalized provider diagnostic, if
// present. Its value is untrusted and may contain sensitive data.
func (e *GenerationError) ProviderMessage() string {
if e == nil {
return ""
}
return e.providerMessage
}
// Error returns a redacted diagnostic that is not a parsing contract.
func (e *GenerationError) Error() string {
if e == nil || e.statusCode == 0 {
return ErrLLMGenerate.Error()
}
return fmt.Sprintf("%s: provider returned HTTP status %d", ErrLLMGenerate, e.statusCode)
}
// GoString returns the same redacted diagnostic as Error.
func (e *GenerationError) GoString() string {
return e.Error()
}
// Unwrap returns ErrLLMGenerate. It is safe to call on a nil receiver or zero
// value.
func (e *GenerationError) Unwrap() error {
return ErrLLMGenerate
}

View File

@@ -0,0 +1,150 @@
package promptkit_test
import (
"context"
"errors"
"fmt"
"io"
"net/http"
"strings"
"testing"
"gitea.maximumdirect.net/eric/promptkit"
)
func TestBuiltInGenerationError(t *testing.T) {
const (
codeMarker = "provider-code-marker"
typeMarker = "provider-type-marker"
messageMarker = "provider-message-marker"
)
engine := newBuiltInGenerationErrorEngine(t, http.StatusUnprocessableEntity,
`{"error":{"code":"`+codeMarker+`","type":"`+typeMarker+`","message":"`+messageMarker+`"}}`)
result, err := engine.Run(context.Background(), generationErrorRunRequest())
if result != nil {
t.Fatalf("Run result = %#v, want nil", result)
}
assertGenerationError(t, err, http.StatusUnprocessableEntity, codeMarker, typeMarker, messageMarker)
preparedEngine := newBuiltInGenerationErrorEngine(t, http.StatusServiceUnavailable, `{"error":{}}`)
prepared, err := preparedEngine.PrepareExecution(context.Background(), generationErrorRunRequest())
if err != nil {
t.Fatalf("PrepareExecution: %v", err)
}
result, err = preparedEngine.RunPrepared(context.Background(), prepared)
if result != nil {
t.Fatalf("RunPrepared result = %#v, want nil", result)
}
assertGenerationError(t, err, http.StatusServiceUnavailable, "", "", "")
}
func TestBuiltInGenerationErrorWithAppendedMessages(t *testing.T) {
const messageMarker = "combined-request-provider-marker"
engine := newBuiltInGenerationErrorEngine(t, http.StatusUnprocessableEntity,
`{"error":{"message":"`+messageMarker+`"}}`)
request := generationErrorRunRequest()
request.AppendedMessages = []promptkit.RenderedMessage{
{Role: promptkit.RoleAssistant, Content: "previous response"},
{Role: promptkit.RoleUser, Content: "consumer correction"},
}
result, err := engine.Run(context.Background(), request)
if result != nil {
t.Fatalf("Run result = %#v, want nil", result)
}
assertGenerationError(t, err, http.StatusUnprocessableEntity, "", "", messageMarker)
}
func TestBuiltInRepairGenerationError(t *testing.T) {
const (
codeMarker = "repair-code-marker"
typeMarker = "repair-type-marker"
messageMarker = "repair-message-marker"
)
calls := 0
config := contractConfig(frameworkSchemaDir)
config.HTTPClient = &http.Client{Transport: roundTripFunc(func(*http.Request) (*http.Response, error) {
calls++
if calls == 1 {
body := `{"choices":[{"message":{"content":"not-json"}}]}`
return &http.Response{StatusCode: http.StatusOK, ContentLength: int64(len(body)), Body: io.NopCloser(strings.NewReader(body))}, nil
}
body := `{"error":{"code":"` + codeMarker + `","type":"` + typeMarker + `","message":"` + messageMarker + `"}}`
return &http.Response{StatusCode: http.StatusUnprocessableEntity, ContentLength: int64(len(body)), Body: io.NopCloser(strings.NewReader(body))}, nil
})}
engine, err := promptkit.NewEngine(config)
if err != nil {
t.Fatalf("NewEngine: %v", err)
}
req := generationErrorRunRequest()
req.Validation = &promptkit.OutputContract{
Format: promptkit.FormatJSON,
ValidationMode: promptkit.ValidationJSON,
RepairAttempts: 1,
}
result, err := engine.Run(context.Background(), req)
if result != nil {
t.Fatalf("Run result = %#v, want nil", result)
}
if calls != 2 {
t.Fatalf("provider calls = %d, want 2", calls)
}
assertGenerationError(t, err, http.StatusUnprocessableEntity, codeMarker, typeMarker, messageMarker)
}
func assertGenerationError(t *testing.T, err error, statusCode int, code, providerType, message string) {
t.Helper()
if !errors.Is(err, promptkit.ErrLLMGenerate) {
t.Fatalf("errors.Is(%v, ErrLLMGenerate) = false", err)
}
var generationErr *promptkit.GenerationError
if !errors.As(err, &generationErr) || generationErr == nil {
t.Fatalf("error = %T, want *GenerationError", err)
}
if generationErr.StatusCode() != statusCode || generationErr.ProviderCode() != code || generationErr.ProviderType() != providerType || generationErr.ProviderMessage() != message {
t.Fatalf("GenerationError = %#v", generationErr)
}
wantFormatted := fmt.Sprintf("failed to generate output: provider returned HTTP status %d", statusCode)
for _, rendered := range []string{fmt.Sprintf("%v", generationErr), fmt.Sprintf("%+v", generationErr), fmt.Sprintf("%#v", generationErr)} {
if rendered != wantFormatted {
t.Fatalf("formatted error = %q, want %q", rendered, wantFormatted)
}
for _, marker := range []string{code, providerType, message} {
if marker != "" && strings.Contains(rendered, marker) {
t.Fatalf("formatted error exposed provider marker %q: %q", marker, rendered)
}
}
}
}
func newBuiltInGenerationErrorEngine(t *testing.T, statusCode int, body string) *promptkit.Engine {
t.Helper()
config := contractConfig(frameworkSchemaDir)
config.HTTPClient = &http.Client{Transport: roundTripFunc(func(*http.Request) (*http.Response, error) {
return &http.Response{
StatusCode: statusCode,
ContentLength: int64(len(body)),
Body: io.NopCloser(strings.NewReader(body)),
}, nil
})}
engine, err := promptkit.NewEngine(config)
if err != nil {
t.Fatalf("NewEngine: %v", err)
}
return engine
}
func generationErrorRunRequest() promptkit.RunRequest {
return promptkit.RunRequest{
PromptID: frameworkMarkdownSummaryPromptID,
Inputs: map[string]promptkit.ArtifactRef{
"transcript": promptkit.Inline("Rin opens the gate."),
"glossary": promptkit.Inline("gate: A guarded passage."),
},
}
}

View File

@@ -0,0 +1,32 @@
package promptkit
import (
"errors"
"fmt"
"testing"
)
func TestGenerationErrorNilAndZeroValue(t *testing.T) {
var nilError *GenerationError
zeroError := &GenerationError{}
for name, err := range map[string]*GenerationError{
"nil": nilError,
"zero": zeroError,
} {
t.Run(name, func(t *testing.T) {
if err.StatusCode() != 0 || err.ProviderCode() != "" || err.ProviderType() != "" || err.ProviderMessage() != "" {
t.Fatalf("accessors returned provider details: %#v", err)
}
if err.Error() != "failed to generate output" || err.GoString() != "failed to generate output" {
t.Fatalf("redacted formatting = (%q, %q)", err.Error(), err.GoString())
}
if fmt.Sprintf("%v", err) != "failed to generate output" || fmt.Sprintf("%#v", err) != "failed to generate output" {
t.Fatalf("formatted error = (%q, %q)", fmt.Sprintf("%v", err), fmt.Sprintf("%#v", err))
}
if !errors.Is(err, ErrLLMGenerate) {
t.Fatalf("errors.Is(%v, ErrLLMGenerate) = false", err)
}
})
}
}

2
go.mod
View File

@@ -3,6 +3,8 @@ module gitea.maximumdirect.net/eric/promptkit
go 1.25.5 go 1.25.5
require ( require (
gitea.maximumdirect.net/eric/promptkit-backend-openrouter v1.0.0
gitea.maximumdirect.net/eric/promptkit-backend-rakestrawhome v1.0.0
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2 github.com/santhosh-tekuri/jsonschema/v6 v6.0.2
gopkg.in/yaml.v3 v3.0.1 gopkg.in/yaml.v3 v3.0.1
) )

4
go.sum
View File

@@ -1,3 +1,7 @@
gitea.maximumdirect.net/eric/promptkit-backend-openrouter v1.0.0 h1:lc062euk2qseO//D762i3JaFyulDNML3eQQX7DkYTho=
gitea.maximumdirect.net/eric/promptkit-backend-openrouter v1.0.0/go.mod h1:AIa7kAu2mfrRQgcspe4L+DW51WqgnALQT60lqkEywJI=
gitea.maximumdirect.net/eric/promptkit-backend-rakestrawhome v1.0.0 h1:j9YY7wsTVjzke2kHH4YAzpU0oUpM+x+nXwl1IeS+2eg=
gitea.maximumdirect.net/eric/promptkit-backend-rakestrawhome v1.0.0/go.mod h1:4RNS+LILDg4JbS4Ts9Lwy1C92wauXJIbeQaalps4Koo=
github.com/dlclark/regexp2 v1.11.0 h1:G/nrcoOa7ZXlpoa/91N3X7mM3r8eIlMBBJZvsz/mxKI= github.com/dlclark/regexp2 v1.11.0 h1:G/nrcoOa7ZXlpoa/91N3X7mM3r8eIlMBBJZvsz/mxKI=
github.com/dlclark/regexp2 v1.11.0/go.mod h1:DHkYz0B9wPfa6wondMfaivmHpzrQ3v9q8cnmRbL6yW8= github.com/dlclark/regexp2 v1.11.0/go.mod h1:DHkYz0B9wPfa6wondMfaivmHpzrQ3v9q8cnmRbL6yW8=
github.com/santhosh-tekuri/jsonschema/v6 v6.0.2 h1:KRzFb2m7YtdldCEkzs6KqmJw4nqEVZGK7IN2kJkjTuQ= github.com/santhosh-tekuri/jsonschema/v6 v6.0.2 h1:KRzFb2m7YtdldCEkzs6KqmJw4nqEVZGK7IN2kJkjTuQ=

View File

@@ -16,10 +16,12 @@ import (
var ( var (
ErrUnsupportedRefType = errors.New("unsupported artifact reference type") ErrUnsupportedRefType = errors.New("unsupported artifact reference type")
ErrMissingInlineBody = errors.New("missing body for inline artifact")
ErrMissingFilePath = errors.New("missing file path for file artifact") ErrMissingFilePath = errors.New("missing file path for file artifact")
ErrUnsupportedFile = errors.New("file artifact path is not a regular file")
) )
const fileReadChunkSize = 64 * 1024
// Reader resolves artifact references into actual artifacts. // Reader resolves artifact references into actual artifacts.
type Reader interface { type Reader interface {
Read(ctx context.Context, ref domain.ArtifactRef) (*domain.Artifact, error) Read(ctx context.Context, ref domain.ArtifactRef) (*domain.Artifact, error)
@@ -34,7 +36,7 @@ type CompositeReader struct {
func NewCompositeReader() Reader { func NewCompositeReader() Reader {
return &CompositeReader{ return &CompositeReader{
inlineReader: &inlineReader{}, inlineReader: &inlineReader{},
fileReader: &fileReader{}, fileReader: &fileReader{open: openArtifactFile},
} }
} }
@@ -64,10 +66,6 @@ func (r *inlineReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domai
default: default:
} }
if ref.Body == "" {
return nil, ErrMissingInlineBody
}
body := []byte(ref.Body) body := []byte(ref.Body)
return &domain.Artifact{ return &domain.Artifact{
ContentType: defaults.ContentTypeTextPlain, ContentType: defaults.ContentTypeTextPlain,
@@ -78,7 +76,15 @@ func (r *inlineReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domai
}, nil }, nil
} }
type fileReader struct{} type artifactFile interface {
Read([]byte) (int, error)
Stat() (os.FileInfo, error)
Close() error
}
type fileReader struct {
open func(string) (artifactFile, error)
}
func (r *fileReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domain.Artifact, error) { func (r *fileReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domain.Artifact, error) {
select { select {
@@ -91,25 +97,71 @@ func (r *fileReader) Read(ctx context.Context, ref domain.ArtifactRef) (*domain.
return nil, ErrMissingFilePath return nil, ErrMissingFilePath
} }
return readFileArtifact(ref.URI) return readFileArtifact(ctx, ref.URI, r.open)
} }
func readFileArtifact(path string) (*domain.Artifact, error) { func openArtifactFile(path string) (artifactFile, error) {
file, err := os.Open(path) return os.Open(path)
}
func readFileArtifact(ctx context.Context, path string, open func(string) (artifactFile, error)) (*domain.Artifact, error) {
if err := ctx.Err(); err != nil {
return nil, err
}
info, err := os.Stat(path)
if err != nil {
return nil, fmt.Errorf("failed to read file %s: %w", path, err)
}
if !info.Mode().IsRegular() {
return nil, fmt.Errorf("%w: %s", ErrUnsupportedFile, path)
}
if err := ctx.Err(); err != nil {
return nil, err
}
file, err := open(path)
if err != nil { if err != nil {
return nil, fmt.Errorf("failed to read file %s: %w", path, err) return nil, fmt.Errorf("failed to read file %s: %w", path, err)
} }
defer file.Close() defer file.Close()
data, err := io.ReadAll(file) openedInfo, err := file.Stat()
if err != nil { if err != nil {
return nil, fmt.Errorf("failed to read file %s: %w", path, err) return nil, fmt.Errorf("failed to inspect opened file %s: %w", path, err)
}
if !openedInfo.Mode().IsRegular() {
return nil, fmt.Errorf("%w: %s", ErrUnsupportedFile, path)
}
data := make([]byte, 0)
chunk := make([]byte, fileReadChunkSize)
for {
if err := ctx.Err(); err != nil {
return nil, err
}
n, readErr := file.Read(chunk)
if n > 0 {
data = append(data, chunk[:n]...)
}
if err := ctx.Err(); err != nil {
return nil, err
}
if errors.Is(readErr, io.EOF) {
break
}
if readErr != nil {
return nil, fmt.Errorf("failed to read file %s: %w", path, readErr)
}
} }
contentType := mime.TypeByExtension(filepath.Ext(path)) contentType := mime.TypeByExtension(filepath.Ext(path))
if contentType == "" { if contentType == "" {
contentType = defaults.ContentTypeTextPlain contentType = defaults.ContentTypeTextPlain
} }
hash := fmt.Sprintf("%x", sha256.Sum256(data))
if err := ctx.Err(); err != nil {
return nil, err
}
return &domain.Artifact{ return &domain.Artifact{
Name: filepath.Base(path), Name: filepath.Base(path),
@@ -117,6 +169,6 @@ func readFileArtifact(path string) (*domain.Artifact, error) {
Body: data, Body: data,
URI: path, URI: path,
Size: int64(len(data)), Size: int64(len(data)),
Hash: fmt.Sprintf("%x", sha256.Sum256(data)), Hash: hash,
}, nil }, nil
} }

View File

@@ -0,0 +1,43 @@
//go:build linux
package artifact
import (
"context"
"errors"
"path/filepath"
"syscall"
"testing"
"time"
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
)
func TestFileReaderRejectsFIFOBeforeOpen(t *testing.T) {
path := filepath.Join(t.TempDir(), "artifact.fifo")
if err := syscall.Mkfifo(path, 0o600); err != nil {
t.Fatalf("create fifo: %v", err)
}
type result struct {
artifact *domain.Artifact
err error
}
done := make(chan result, 1)
go func() {
artifact, err := NewCompositeReader().Read(context.Background(), domain.ArtifactRef{
Type: domain.ArtifactRefFile,
URI: path,
})
done <- result{artifact: artifact, err: err}
}()
select {
case got := <-done:
if got.artifact != nil || !errors.Is(got.err, ErrUnsupportedFile) {
t.Fatalf("artifact=%#v err=%v, want nil/ErrUnsupportedFile", got.artifact, got.err)
}
case <-time.After(time.Second):
t.Fatal("FIFO read blocked instead of rejecting the non-regular file")
}
}

View File

@@ -1,6 +1,7 @@
package artifact package artifact
import ( import (
"bytes"
"context" "context"
"errors" "errors"
"os" "os"
@@ -11,54 +12,97 @@ import (
"gitea.maximumdirect.net/eric/promptkit/internal/domain" "gitea.maximumdirect.net/eric/promptkit/internal/domain"
) )
func TestCompositeReader_Read(t *testing.T) { func TestCompositeReaderRejectsUnsupportedReferences(t *testing.T) {
_, err := NewCompositeReader().Read(context.Background(), domain.ArtifactRef{
Type: domain.ArtifactRefType("unsupported"),
URI: "unsupported://bucket/key",
})
if !errors.Is(err, ErrUnsupportedRefType) {
t.Fatalf("expected ErrUnsupportedRefType, got %v", err)
}
}
func TestCompositeReaderSourceParityAndOpaqueHashes(t *testing.T) {
reader := NewCompositeReader() reader := NewCompositeReader()
ctx := context.Background() hashes := make(map[string]string)
tests := []struct {
name string
content string
}{
{name: "empty", content: ""},
{name: "ordinary", content: "same content"},
{name: "changed", content: "changed content"},
}
t.Run("inline artifact", func(t *testing.T) { for _, tc := range tests {
ref := domain.ArtifactRef{ t.Run(tc.name, func(t *testing.T) {
Type: domain.ArtifactRefInline, filePath := filepath.Join(t.TempDir(), "artifact.txt")
Body: "hello world", if err := os.WriteFile(filePath, []byte(tc.content), 0o600); err != nil {
} t.Fatal(err)
art, err := reader.Read(ctx, ref) }
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if string(art.Body) != "hello world" {
t.Errorf("expected 'hello world', got %s", string(art.Body))
}
if art.ContentType != "text/plain" {
t.Errorf("expected text/plain content type, got %q", art.ContentType)
}
if art.Hash != "b94d27b9934d3e08a52e52d7da7dabfac484efe37a5380ee9088f7ace2efcde9" {
t.Errorf("unexpected hash: %s", art.Hash)
}
if art.Size != int64(len(ref.Body)) {
t.Errorf("expected size %d, got %d", len(ref.Body), art.Size)
}
})
t.Run("inline artifact missing body", func(t *testing.T) { sources := []struct {
ref := domain.ArtifactRef{ name string
Type: domain.ArtifactRefInline, ref domain.ArtifactRef
Body: "", wantURI string
} }{
_, err := reader.Read(ctx, ref) {
if !errors.Is(err, ErrMissingInlineBody) { name: "inline",
t.Errorf("expected ErrMissingInlineBody, got %v", err) ref: domain.ArtifactRef{Type: domain.ArtifactRefInline, Body: tc.content},
} },
}) {
name: "inline with uri",
ref: domain.ArtifactRef{Type: domain.ArtifactRefInline, URI: "memory://input", Body: tc.content},
wantURI: "memory://input",
},
{
name: "file",
ref: domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: filePath},
wantURI: filePath,
},
}
t.Run("unsupported ref type", func(t *testing.T) { var sourceHash string
ref := domain.ArtifactRef{ for _, source := range sources {
Type: domain.ArtifactRefType("unsupported"), t.Run(source.name, func(t *testing.T) {
URI: "unsupported://bucket/key", first, err := reader.Read(context.Background(), source.ref)
} if err != nil {
_, err := reader.Read(ctx, ref) t.Fatalf("first read: %v", err)
if !errors.Is(err, ErrUnsupportedRefType) { }
t.Error("expected error for unsupported type") second, err := reader.Read(context.Background(), source.ref)
} if err != nil {
}) t.Fatalf("second read: %v", err)
}
if string(first.Body) != tc.content || first.Size != int64(len(tc.content)) {
t.Fatalf("body=%q size=%d, want %q/%d", first.Body, first.Size, tc.content, len(tc.content))
}
if first.URI != source.wantURI {
t.Fatalf("URI = %q, want %q", first.URI, source.wantURI)
}
if first.Hash == "" || first.Hash != second.Hash {
t.Fatalf("hashes are not non-empty and stable: %q/%q", first.Hash, second.Hash)
}
if sourceHash == "" {
sourceHash = first.Hash
} else if first.Hash != sourceHash {
t.Fatalf("equal content hashes differ: %q/%q", sourceHash, first.Hash)
}
if source.ref.Type == domain.ArtifactRefFile {
if first.Name != filepath.Base(filePath) || !strings.HasPrefix(first.ContentType, "text/plain") {
t.Fatalf("unexpected file metadata: %+v", first)
}
} else if first.ContentType != "text/plain" {
t.Fatalf("inline content type = %q", first.ContentType)
}
})
}
hashes[tc.name] = sourceHash
})
}
if hashes["empty"] == hashes["ordinary"] || hashes["ordinary"] == hashes["changed"] {
t.Fatalf("changed content did not change opaque hash: %#v", hashes)
}
} }
func TestCompositeReaderCopiesInlineData(t *testing.T) { func TestCompositeReaderCopiesInlineData(t *testing.T) {
@@ -87,94 +131,154 @@ func TestCompositeReaderCopiesInlineData(t *testing.T) {
} }
} }
func TestCompositeReaderHonorsCancellation(t *testing.T) { func TestCompositeReaderHonorsPreCancellation(t *testing.T) {
ctx, cancel := context.WithCancel(context.Background()) filePath := filepath.Join(t.TempDir(), "artifact.txt")
cancel() if err := os.WriteFile(filePath, []byte("ignored"), 0o600); err != nil {
t.Fatal(err)
}
tests := []struct {
name string
ref domain.ArtifactRef
}{
{name: "inline", ref: domain.ArtifactRef{Type: domain.ArtifactRefInline, Body: "ignored"}},
{name: "file", ref: domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: filePath}},
}
_, err := NewCompositeReader().Read(ctx, domain.ArtifactRef{ for _, tc := range tests {
Type: domain.ArtifactRefInline, t.Run(tc.name, func(t *testing.T) {
Body: "ignored", ctx, cancel := context.WithCancel(context.Background())
}) cancel()
if !errors.Is(err, context.Canceled) {
t.Fatalf("expected context cancellation, got %v", err) artifact, err := NewCompositeReader().Read(ctx, tc.ref)
if artifact != nil || !errors.Is(err, context.Canceled) {
t.Fatalf("artifact=%#v err=%v, want nil/context.Canceled", artifact, err)
}
})
} }
} }
func TestFileReader_Read(t *testing.T) { func TestFileReaderFailuresAndMetadata(t *testing.T) {
content := []byte("test file content")
filePath := filepath.Join(t.TempDir(), "artifact.txt")
if err := os.WriteFile(filePath, content, 0o600); err != nil {
t.Fatal(err)
}
reader := NewCompositeReader() reader := NewCompositeReader()
ctx := context.Background()
t.Run("file artifact loading", func(t *testing.T) {
ref := domain.ArtifactRef{
Type: domain.ArtifactRefFile,
URI: filePath,
}
art, err := reader.Read(ctx, ref)
if err != nil {
t.Fatalf("unexpected error: %v", err)
}
if string(art.Body) != string(content) {
t.Errorf("expected %s, got %s", string(content), string(art.Body))
}
if art.Name != filepath.Base(filePath) {
t.Errorf("expected name %q, got %q", filepath.Base(filePath), art.Name)
}
if !strings.HasPrefix(art.ContentType, "text/plain") {
t.Errorf("expected text content type, got %q", art.ContentType)
}
if art.URI != filePath {
t.Errorf("expected URI %q, got %q", filePath, art.URI)
}
if art.Size != int64(len(content)) {
t.Errorf("expected size %d, got %d", len(content), art.Size)
}
if art.Hash != "60f5237ed4049f0382661ef009d2bc42e48c3ceb3edb6600f7024e7ab3b838f3" {
t.Errorf("unexpected hash: %s", art.Hash)
}
})
t.Run("missing file path", func(t *testing.T) { t.Run("missing file path", func(t *testing.T) {
ref := domain.ArtifactRef{ _, err := reader.Read(context.Background(), domain.ArtifactRef{Type: domain.ArtifactRefFile})
Type: domain.ArtifactRefFile,
URI: "",
}
_, err := reader.Read(ctx, ref)
if !errors.Is(err, ErrMissingFilePath) { if !errors.Is(err, ErrMissingFilePath) {
t.Errorf("expected ErrMissingFilePath, got %v", err) t.Fatalf("expected ErrMissingFilePath, got %v", err)
} }
}) })
t.Run("missing file", func(t *testing.T) { t.Run("missing file", func(t *testing.T) {
ref := domain.ArtifactRef{ _, err := reader.Read(context.Background(), domain.ArtifactRef{
Type: domain.ArtifactRefFile, Type: domain.ArtifactRefFile,
URI: filepath.Join(t.TempDir(), "missing.txt"), URI: filepath.Join(t.TempDir(), "missing.txt"),
} })
if _, err := reader.Read(ctx, ref); err == nil { if err == nil {
t.Fatal("expected missing file error") t.Fatal("expected missing file error")
} }
}) })
t.Run("directory rejected before open", func(t *testing.T) {
artifact, err := reader.Read(context.Background(), domain.ArtifactRef{
Type: domain.ArtifactRefFile,
URI: t.TempDir(),
})
if artifact != nil || !errors.Is(err, ErrUnsupportedFile) {
t.Fatalf("artifact=%#v err=%v, want nil/ErrUnsupportedFile", artifact, err)
}
})
t.Run("non-regular opened target rejected", func(t *testing.T) {
filePath := filepath.Join(t.TempDir(), "artifact.txt")
if err := os.WriteFile(filePath, []byte("content"), 0o600); err != nil {
t.Fatal(err)
}
directoryInfo, err := os.Stat(t.TempDir())
if err != nil {
t.Fatal(err)
}
fileReader := &fileReader{open: func(path string) (artifactFile, error) {
file, err := os.Open(path)
if err != nil {
return nil, err
}
return &reportedInfoFile{artifactFile: file, info: directoryInfo}, nil
}}
artifact, err := fileReader.Read(context.Background(), domain.ArtifactRef{
Type: domain.ArtifactRefFile,
URI: filePath,
})
if artifact != nil || !errors.Is(err, ErrUnsupportedFile) {
t.Fatalf("artifact=%#v err=%v, want nil/ErrUnsupportedFile", artifact, err)
}
})
t.Run("unknown extension uses text fallback", func(t *testing.T) { t.Run("unknown extension uses text fallback", func(t *testing.T) {
path := filepath.Join(t.TempDir(), "artifact.unknownextension") filePath := filepath.Join(t.TempDir(), "artifact.unknownextension")
if err := os.WriteFile(path, content, 0o600); err != nil { if err := os.WriteFile(filePath, []byte("content"), 0o600); err != nil {
t.Fatal(err) t.Fatal(err)
} }
art, err := reader.Read(ctx, domain.ArtifactRef{ artifact, err := reader.Read(context.Background(), domain.ArtifactRef{
Type: domain.ArtifactRefFile, Type: domain.ArtifactRefFile,
URI: path, URI: filePath,
}) })
if err != nil { if err != nil {
t.Fatalf("unexpected error: %v", err) t.Fatalf("read artifact: %v", err)
} }
if art.ContentType != "text/plain" { if artifact.ContentType != "text/plain" {
t.Errorf("expected text/plain fallback, got %q", art.ContentType) t.Fatalf("content type = %q", artifact.ContentType)
} }
}) })
} }
func TestFileReaderCancelsAfterReadProgress(t *testing.T) {
filePath := filepath.Join(t.TempDir(), "artifact.bin")
content := bytes.Repeat([]byte("x"), fileReadChunkSize*2)
if err := os.WriteFile(filePath, content, 0o600); err != nil {
t.Fatal(err)
}
ctx, cancel := context.WithCancel(context.Background())
var opened *cancelAfterProgressFile
reader := &fileReader{open: func(path string) (artifactFile, error) {
file, err := os.Open(path)
if err != nil {
return nil, err
}
opened = &cancelAfterProgressFile{artifactFile: file, cancel: cancel}
return opened, nil
}}
artifact, err := reader.Read(ctx, domain.ArtifactRef{Type: domain.ArtifactRefFile, URI: filePath})
if artifact != nil || !errors.Is(err, context.Canceled) {
t.Fatalf("artifact=%#v err=%v, want nil/context.Canceled", artifact, err)
}
if opened == nil || opened.reads != 1 {
t.Fatalf("read count = %v, want one progressing read", opened)
}
}
type reportedInfoFile struct {
artifactFile
info os.FileInfo
}
func (f *reportedInfoFile) Stat() (os.FileInfo, error) {
return f.info, nil
}
type cancelAfterProgressFile struct {
artifactFile
cancel context.CancelFunc
reads int
}
func (f *cancelAfterProgressFile) Read(buffer []byte) (int, error) {
n, err := f.artifactFile.Read(buffer)
if n > 0 {
f.reads++
f.cancel()
}
return n, err
}

View File

@@ -5,7 +5,6 @@ package backend
import ( import (
"errors" "errors"
"fmt" "fmt"
"net/url"
"regexp" "regexp"
"sort" "sort"
"strings" "strings"
@@ -19,12 +18,11 @@ const (
// OpenRouterID is the reserved ID of Promptkit's built-in OpenRouter // OpenRouterID is the reserved ID of Promptkit's built-in OpenRouter
// backend. // backend.
OpenRouterID = "openrouter" OpenRouterID = "openrouter"
// RakestrawHomeID is the reserved ID of Promptkit's built-in Rakestrawhome
// backend.
RakestrawHomeID = "rakestrawhome"
openRouterEndpoint = "https://openrouter.ai/api/v1" defaultQueueCapacity = 1024
openRouterAPIKeyEnv = "OPENROUTER_API_KEY"
openRouterConcurrencyLimit = 16
defaultQueueCapacity = 1024
) )
// ErrBackendNotFound identifies a registry lookup for an unknown backend ID. // ErrBackendNotFound identifies a registry lookup for an unknown backend ID.
@@ -37,20 +35,15 @@ type Registry struct {
backends map[string]domain.Backend backends map[string]domain.Backend
} }
// NewRegistry constructs a registry containing the built-in OpenRouter // NewRegistry constructs a registry containing maintained definitions followed
// definition followed by the supplied additions. Every ID must be unique. // by consumer additions. Every ID must be unique across both groups.
func NewRegistry(additions []domain.Backend) (*Registry, error) { func NewRegistry(maintained, additions []domain.Backend) (*Registry, error) {
registry := &Registry{ registry := &Registry{
backends: make(map[string]domain.Backend, len(additions)+1), backends: make(map[string]domain.Backend, len(maintained)+len(additions)),
} }
definitions := make([]domain.Backend, 0, len(additions)+1) definitions := make([]domain.Backend, 0, len(maintained)+len(additions))
definitions = append(definitions, domain.Backend{ definitions = append(definitions, maintained...)
ID: OpenRouterID,
Endpoint: openRouterEndpoint,
APIKeyEnv: openRouterAPIKeyEnv,
ConcurrencyLimit: openRouterConcurrencyLimit,
})
definitions = append(definitions, additions...) definitions = append(definitions, additions...)
for _, definition := range definitions { for _, definition := range definitions {
@@ -62,7 +55,7 @@ func NewRegistry(additions []domain.Backend) (*Registry, error) {
return nil, fmt.Errorf("backend ID %q is already registered", definition.ID) return nil, fmt.Errorf("backend ID %q is already registered", definition.ID)
} }
normalized, err := normalizeBackend(definition) normalized, err := NormalizeDefinition(definition)
if err != nil { if err != nil {
return nil, err return nil, err
} }
@@ -108,11 +101,13 @@ func (r *Registry) CapacityPolicies() map[string]domain.BackendCapacityPolicy {
return policies return policies
} }
func normalizeBackend(definition domain.Backend) (domain.Backend, error) { // NormalizeDefinition validates and defensively copies one backend definition.
definition.Endpoint = strings.TrimSpace(definition.Endpoint) func NormalizeDefinition(definition domain.Backend) (domain.Backend, error) {
if err := validateEndpoint(definition.Endpoint); err != nil { endpoint, err := domain.NormalizeOpenAICompatibleBaseEndpoint(definition.Endpoint)
if err != nil {
return domain.Backend{}, fmt.Errorf("backend %q endpoint: %w", definition.ID, err) return domain.Backend{}, fmt.Errorf("backend %q endpoint: %w", definition.ID, err)
} }
definition.Endpoint = endpoint
definition.APIKeyEnv = strings.TrimSpace(definition.APIKeyEnv) definition.APIKeyEnv = strings.TrimSpace(definition.APIKeyEnv)
if definition.APIKeyEnv != "" && !environmentVariableName.MatchString(definition.APIKeyEnv) { if definition.APIKeyEnv != "" && !environmentVariableName.MatchString(definition.APIKeyEnv) {
@@ -182,31 +177,3 @@ func normalizeBackend(definition domain.Backend) (domain.Backend, error) {
definition.ExtraParams = extraParams definition.ExtraParams = extraParams
return definition, nil return definition, nil
} }
func validateEndpoint(endpoint string) error {
if endpoint == "" {
return errors.New("must not be blank")
}
if strings.Contains(endpoint, "#") {
return errors.New("must not contain a fragment")
}
parsed, err := url.Parse(endpoint)
if err != nil {
return fmt.Errorf("must be a valid URL: %w", err)
}
scheme := strings.ToLower(parsed.Scheme)
if scheme != "http" && scheme != "https" {
return errors.New("must use http or https")
}
if !parsed.IsAbs() || parsed.Hostname() == "" {
return errors.New("must be absolute and include a host")
}
if parsed.User != nil {
return errors.New("must not contain user information")
}
if parsed.RawQuery != "" || parsed.ForceQuery {
return errors.New("must not contain a query string")
}
return nil
}

View File

@@ -2,6 +2,7 @@ package backend_test
import ( import (
"errors" "errors"
"reflect"
"strings" "strings"
"testing" "testing"
@@ -11,32 +12,39 @@ import (
const validEndpoint = "https://backend.example/v1" const validEndpoint = "https://backend.example/v1"
func TestRegistryIncludesExactOpenRouterDefinition(t *testing.T) { func TestRegistryIncludesMaintainedDefinitions(t *testing.T) {
registry, err := backend.NewRegistry(nil) maintained := []domain.Backend{
{ID: backend.OpenRouterID, Endpoint: validEndpoint, ConcurrencyLimit: 2, QueueCapacity: 3, QueueCapacitySet: true},
{ID: backend.RakestrawHomeID, Endpoint: "https://second.example/v1", ConcurrencyLimit: 4, QueueCapacity: 5, QueueCapacitySet: true},
}
registry, err := backend.NewRegistry(maintained, nil)
if err != nil { if err != nil {
t.Fatalf("construct registry: %v", err) t.Fatalf("construct registry: %v", err)
} }
definition, err := registry.GetBackend(backend.OpenRouterID) for _, expected := range maintained {
if err != nil { t.Run(expected.ID, func(t *testing.T) {
t.Fatalf("look up OpenRouter: %v", err) definition, err := registry.GetBackend(expected.ID)
} if err != nil {
if definition.ID != "openrouter" || t.Fatalf("look up maintained definition: %v", err)
definition.Endpoint != "https://openrouter.ai/api/v1" || }
definition.APIKeyEnv != "OPENROUTER_API_KEY" || if !reflect.DeepEqual(definition, expected) {
definition.ConcurrencyLimit != 16 || t.Fatalf("unexpected maintained definition: %#v", definition)
definition.QueueCapacity != 1024 || }
!definition.QueueCapacitySet || })
definition.ExtraParams != nil {
t.Fatalf("unexpected OpenRouter definition: %#v", definition)
} }
policies := registry.CapacityPolicies() policies := registry.CapacityPolicies()
if len(policies) != 1 || if len(policies) != 2 ||
policies["openrouter"] != (domain.BackendCapacityPolicy{ policies[backend.OpenRouterID] != (domain.BackendCapacityPolicy{
ConcurrencyLimit: 16, ConcurrencyLimit: 2,
QueueCapacity: 1024, QueueCapacity: 3,
}) ||
policies[backend.RakestrawHomeID] != (domain.BackendCapacityPolicy{
ConcurrencyLimit: 4,
QueueCapacity: 5,
}) { }) {
t.Fatalf("unexpected OpenRouter capacity policies: %#v", policies) t.Fatalf("unexpected built-in capacity policies: %#v", policies)
} }
} }
@@ -46,7 +54,7 @@ func TestRegistryNormalizesUniqueAdditionsAndIsolatesMutations(t *testing.T) {
"count": int64(7), "count": int64(7),
"nested": nested, "nested": nested,
} }
registry, err := backend.NewRegistry([]domain.Backend{ registry, err := backend.NewRegistry(nil, []domain.Backend{
{ {
ID: " custom ", ID: " custom ",
Endpoint: " https://custom.example/openai/v1 ", Endpoint: " https://custom.example/openai/v1 ",
@@ -109,11 +117,11 @@ func TestRegistryNormalizesUniqueAdditionsAndIsolatesMutations(t *testing.T) {
} }
policies := registry.CapacityPolicies() policies := registry.CapacityPolicies()
if len(policies) != 2 { if len(policies) != 1 {
t.Fatalf("unexpected capacity policy count: %#v", policies) t.Fatalf("unexpected capacity policy count: %#v", policies)
} }
policies["custom"] = domain.BackendCapacityPolicy{} policies["custom"] = domain.BackendCapacityPolicy{}
delete(policies, backend.OpenRouterID) delete(policies, "custom")
againPolicies := registry.CapacityPolicies() againPolicies := registry.CapacityPolicies()
if againPolicies["custom"] != (domain.BackendCapacityPolicy{ if againPolicies["custom"] != (domain.BackendCapacityPolicy{
ConcurrencyLimit: 3, ConcurrencyLimit: 3,
@@ -121,9 +129,6 @@ func TestRegistryNormalizesUniqueAdditionsAndIsolatesMutations(t *testing.T) {
}) { }) {
t.Fatalf("capacity policy map mutated registry state: %#v", againPolicies) t.Fatalf("capacity policy map mutated registry state: %#v", againPolicies)
} }
if _, ok := againPolicies[backend.OpenRouterID]; !ok {
t.Fatalf("capacity policy deletion mutated registry state: %#v", againPolicies)
}
} }
func TestNewRegistryNormalizesCapacityPolicy(t *testing.T) { func TestNewRegistryNormalizesCapacityPolicy(t *testing.T) {
@@ -199,7 +204,7 @@ func TestNewRegistryNormalizesCapacityPolicy(t *testing.T) {
t.Run(tc.name, func(t *testing.T) { t.Run(tc.name, func(t *testing.T) {
tc.definition.ID = "custom" tc.definition.ID = "custom"
tc.definition.Endpoint = validEndpoint tc.definition.Endpoint = validEndpoint
registry, err := backend.NewRegistry([]domain.Backend{tc.definition}) registry, err := backend.NewRegistry(nil, []domain.Backend{tc.definition})
if tc.wantError { if tc.wantError {
if err == nil { if err == nil {
t.Fatal("expected invalid capacity policy error") t.Fatal("expected invalid capacity policy error")
@@ -242,11 +247,14 @@ func TestNewRegistryRejectsDuplicateIDs(t *testing.T) {
wantID string wantID string
}{ }{
{ {
name: "built-in collision after normalization", name: "OpenRouter collision after normalization",
additions: []domain.Backend{{ additions: []domain.Backend{{ID: " openrouter "}},
ID: " openrouter ", wantID: backend.OpenRouterID,
}}, },
wantID: "openrouter", {
name: "Rakestrawhome collision after normalization",
additions: []domain.Backend{{ID: " rakestrawhome "}},
wantID: backend.RakestrawHomeID,
}, },
{ {
name: "consumer collision after normalization", name: "consumer collision after normalization",
@@ -260,7 +268,7 @@ func TestNewRegistryRejectsDuplicateIDs(t *testing.T) {
for _, tc := range tests { for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) { t.Run(tc.name, func(t *testing.T) {
_, err := backend.NewRegistry(tc.additions) _, err := backend.NewRegistry(nil, tc.additions)
if err == nil { if err == nil {
t.Fatal("expected duplicate ID error") t.Fatal("expected duplicate ID error")
} }
@@ -274,7 +282,7 @@ func TestNewRegistryRejectsDuplicateIDs(t *testing.T) {
func TestNewRegistryValidatesIDs(t *testing.T) { func TestNewRegistryValidatesIDs(t *testing.T) {
for _, id := range []string{"", " \t\n "} { for _, id := range []string{"", " \t\n "} {
t.Run(id, func(t *testing.T) { t.Run(id, func(t *testing.T) {
_, err := backend.NewRegistry([]domain.Backend{{ _, err := backend.NewRegistry(nil, []domain.Backend{{
ID: id, ID: id,
Endpoint: validEndpoint, Endpoint: validEndpoint,
}}) }})
@@ -303,7 +311,7 @@ func TestNewRegistryValidatesEndpoints(t *testing.T) {
for _, tc := range tests { for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) { t.Run(tc.name, func(t *testing.T) {
_, err := backend.NewRegistry([]domain.Backend{{ _, err := backend.NewRegistry(nil, []domain.Backend{{
ID: "custom", ID: "custom",
Endpoint: tc.endpoint, Endpoint: tc.endpoint,
}}) }})
@@ -317,7 +325,7 @@ func TestNewRegistryValidatesEndpoints(t *testing.T) {
func TestNewRegistryValidatesEnvironmentVariableNames(t *testing.T) { func TestNewRegistryValidatesEnvironmentVariableNames(t *testing.T) {
for _, name := range []string{"1API_KEY", "API-KEY", "API KEY", "ÅPI_KEY"} { for _, name := range []string{"1API_KEY", "API-KEY", "API KEY", "ÅPI_KEY"} {
t.Run(name, func(t *testing.T) { t.Run(name, func(t *testing.T) {
_, err := backend.NewRegistry([]domain.Backend{{ _, err := backend.NewRegistry(nil, []domain.Backend{{
ID: "custom", ID: "custom",
Endpoint: validEndpoint, Endpoint: validEndpoint,
APIKeyEnv: name, APIKeyEnv: name,
@@ -340,7 +348,7 @@ func TestNewRegistryRejectsInvalidAndReservedExtraParameters(t *testing.T) {
for _, tc := range tests { for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) { t.Run(tc.name, func(t *testing.T) {
_, err := backend.NewRegistry([]domain.Backend{{ _, err := backend.NewRegistry(nil, []domain.Backend{{
ID: "custom", ID: "custom",
Endpoint: validEndpoint, Endpoint: validEndpoint,
ExtraParams: tc.extraParams, ExtraParams: tc.extraParams,
@@ -353,7 +361,7 @@ func TestNewRegistryRejectsInvalidAndReservedExtraParameters(t *testing.T) {
} }
func TestRegistryLookupReportsNotFound(t *testing.T) { func TestRegistryLookupReportsNotFound(t *testing.T) {
registry, err := backend.NewRegistry(nil) registry, err := backend.NewRegistry(nil, nil)
if err != nil { if err != nil {
t.Fatalf("construct registry: %v", err) t.Fatalf("construct registry: %v", err)
} }

View File

@@ -124,6 +124,11 @@ func TestManagerAdmissionHonorsContextAndUnlimitedBackends(t *testing.T) {
if err != nil { if err != nil {
t.Fatalf("construct manager: %v", err) t.Fatalf("construct manager: %v", err)
} }
release, err := manager.Admit(context.Background(), "limited")
if err != nil {
t.Fatalf("fill limited pool: %v", err)
}
defer release()
ctx, cancel := context.WithCancel(context.Background()) ctx, cancel := context.WithCancel(context.Background())
cancel() cancel()

246
internal/catalog/catalog.go Normal file
View File

@@ -0,0 +1,246 @@
// Package catalog validates immutable maintained backend catalog assets.
package catalog
import (
"bytes"
"context"
"encoding/json"
"errors"
"fmt"
"io"
"io/fs"
"path"
"sort"
"strings"
"gitea.maximumdirect.net/eric/promptkit/internal/backend"
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
"gitea.maximumdirect.net/eric/promptkit/internal/profile"
)
// Source identifies one immutable backend catalog asset tree.
type Source struct {
Name string
ExpectedBackendID string
FS fs.FS
Root string
}
// Set is the validated maintained backend and raw profile catalog.
type Set struct {
Backends []domain.Backend
Profiles profile.Repository
profileIDs []string
}
// Load validates and combines immutable catalog sources in source order.
func Load(sources ...Source) (Set, error) {
if len(sources) == 0 {
return Set{}, errors.New("at least one catalog source is required")
}
loaded := Set{Backends: make([]domain.Backend, 0, len(sources))}
names := map[string]bool{}
backendIDs := map[string]bool{}
profileIDs := map[string]bool{}
for _, source := range sources {
if err := validateSource(source, names); err != nil {
return Set{}, err
}
names[source.Name] = true
definition, err := loadBackend(source)
if err != nil {
return Set{}, err
}
if backendIDs[definition.ID] {
return Set{}, fmt.Errorf("catalog %s: backend ID duplicates an earlier catalog", source.Name)
}
backendIDs[definition.ID] = true
repository, metadata, err := profile.LoadFSRepository(context.Background(), source.FS, path.Join(source.Root, "profiles"))
if err != nil {
return Set{}, fmt.Errorf("catalog %s: %w", source.Name, err)
}
if len(metadata) == 0 {
return Set{}, fmt.Errorf("catalog %s: profiles must not be empty", source.Name)
}
resolvingRepository := profile.NewResolvingRepository(repository)
for _, entry := range metadata {
if profileIDs[entry.ID] {
return Set{}, fmt.Errorf("catalog %s: %s duplicates an earlier profile ID", source.Name, entry.Path)
}
if containsField(entry.ExplicitFields, "endpoint") || containsField(entry.ExplicitFields, "api_key_env") {
return Set{}, fmt.Errorf("catalog %s: %s contains connection metadata", source.Name, entry.Path)
}
value, err := repository.GetProfile(context.Background(), entry.ID)
if err != nil {
return Set{}, fmt.Errorf("catalog %s: %s: %w", source.Name, entry.Path, err)
}
if err := rejectSecretKeys(value.ExtraParams); err != nil {
return Set{}, fmt.Errorf("catalog %s: %s: prohibited extra parameter key", source.Name, entry.Path)
}
resolved, err := resolvingRepository.GetProfile(context.Background(), entry.ID)
if err != nil {
return Set{}, fmt.Errorf("catalog %s: %s has invalid profile inheritance", source.Name, entry.Path)
}
if resolved.BackendID != definition.ID {
return Set{}, fmt.Errorf("catalog %s: %s selects a different backend", source.Name, entry.Path)
}
profileIDs[entry.ID] = true
loaded.profileIDs = append(loaded.profileIDs, entry.ID)
}
if loaded.Profiles == nil {
loaded.Profiles = repository
} else {
loaded.Profiles = profile.NewOverlayRepository(loaded.Profiles, repository)
}
loaded.Backends = append(loaded.Backends, definition)
}
sort.Strings(loaded.profileIDs)
return loaded, nil
}
func validateSource(source Source, names map[string]bool) error {
if strings.TrimSpace(source.Name) == "" || source.Name != strings.TrimSpace(source.Name) {
return errors.New("catalog source name must not be blank")
}
if names[source.Name] {
return fmt.Errorf("duplicate catalog source name %q", source.Name)
}
if source.FS == nil {
return fmt.Errorf("catalog %s: filesystem is nil", source.Name)
}
if source.Root == "." || source.Root == "" || !fs.ValidPath(source.Root) {
return fmt.Errorf("catalog %s: asset root is invalid", source.Name)
}
if strings.TrimSpace(source.ExpectedBackendID) == "" {
return fmt.Errorf("catalog %s: expected backend ID is blank", source.Name)
}
return validateLayout(source)
}
func validateLayout(source Source) error {
manifestPath := path.Join(source.Root, "backend.json")
profilesRoot := path.Join(source.Root, "profiles")
return fs.WalkDir(source.FS, source.Root, func(assetPath string, entry fs.DirEntry, err error) error {
if err != nil {
return err
}
if assetPath == source.Root {
if !entry.IsDir() {
return fmt.Errorf("catalog %s: invalid asset path %s", source.Name, assetPath)
}
return nil
}
if entry.IsDir() {
if assetPath == profilesRoot || strings.HasPrefix(assetPath, profilesRoot+"/") {
return nil
}
return fmt.Errorf("catalog %s: invalid asset path %s", source.Name, assetPath)
}
if assetPath == manifestPath && entry.Type().IsRegular() {
return nil
}
if strings.HasPrefix(assetPath, profilesRoot+"/") && entry.Type().IsRegular() && strings.HasSuffix(assetPath, ".yml") {
return nil
}
return fmt.Errorf("catalog %s: invalid asset path %s", source.Name, assetPath)
})
}
func loadBackend(source Source) (domain.Backend, error) {
data, err := fs.ReadFile(source.FS, path.Join(source.Root, "backend.json"))
if err != nil {
return domain.Backend{}, fmt.Errorf("catalog %s: backend.json: %w", source.Name, err)
}
type manifest struct {
SchemaVersion *int `json:"schema_version"`
ID *string `json:"id"`
Endpoint *string `json:"endpoint"`
APIKeyEnv *string `json:"api_key_env"`
ConcurrencyLimit *int `json:"concurrency_limit"`
QueueCapacity *int `json:"queue_capacity"`
ExtraParams json.RawMessage `json:"extra_params"`
}
decoder := json.NewDecoder(bytes.NewReader(data))
decoder.DisallowUnknownFields()
var value manifest
if err := decoder.Decode(&value); err != nil {
return domain.Backend{}, fmt.Errorf("catalog %s: backend.json: invalid manifest", source.Name)
}
var trailing any
if err := decoder.Decode(&trailing); !errors.Is(err, io.EOF) {
return domain.Backend{}, fmt.Errorf("catalog %s: backend.json: invalid manifest", source.Name)
}
if value.SchemaVersion == nil || *value.SchemaVersion != 1 || value.ID == nil || value.Endpoint == nil || value.APIKeyEnv == nil || value.ConcurrencyLimit == nil || value.QueueCapacity == nil || value.ExtraParams == nil {
return domain.Backend{}, fmt.Errorf("catalog %s: backend.json: required field is missing or unsupported", source.Name)
}
if *value.ID != source.ExpectedBackendID {
return domain.Backend{}, fmt.Errorf("catalog %s: backend ID does not match expected ID", source.Name)
}
if strings.TrimSpace(*value.APIKeyEnv) == "" {
return domain.Backend{}, fmt.Errorf("catalog %s: backend.json: api key environment variable must not be blank", source.Name)
}
extraParams, err := decodeExtraParams(value.ExtraParams)
if err != nil {
return domain.Backend{}, fmt.Errorf("catalog %s: backend.json: invalid extra parameters", source.Name)
}
if err := rejectSecretKeys(extraParams); err != nil {
return domain.Backend{}, fmt.Errorf("catalog %s: backend.json: prohibited extra parameter key", source.Name)
}
normalized, err := backend.NormalizeDefinition(domain.Backend{ID: *value.ID, Endpoint: *value.Endpoint, APIKeyEnv: *value.APIKeyEnv, ExtraParams: extraParams, ConcurrencyLimit: *value.ConcurrencyLimit, QueueCapacity: *value.QueueCapacity, QueueCapacitySet: true})
if err != nil {
return domain.Backend{}, fmt.Errorf("catalog %s: backend.json: invalid backend definition", source.Name)
}
return normalized, nil
}
func decodeExtraParams(data []byte) (map[string]any, error) {
if bytes.Equal(bytes.TrimSpace(data), []byte("null")) {
return nil, nil
}
decoder := json.NewDecoder(bytes.NewReader(data))
decoder.UseNumber()
var value map[string]any
if err := decoder.Decode(&value); err != nil || value == nil {
return nil, errors.New("extra parameters must be an object or null")
}
var trailing any
if err := decoder.Decode(&trailing); !errors.Is(err, io.EOF) {
return nil, errors.New("extra parameters must contain exactly one JSON value")
}
return value, nil
}
func containsField(fields []string, target string) bool {
for _, field := range fields {
if field == target {
return true
}
}
return false
}
func rejectSecretKeys(value map[string]any) error {
return rejectSecretValue(value)
}
func rejectSecretValue(value any) error {
switch value := value.(type) {
case map[string]any:
for key, child := range value {
switch strings.ToLower(key) {
case "api_key", "apikey", "authorization", "credential", "credentials", "password", "secret", "token", "access_token":
return errors.New("prohibited key")
}
if err := rejectSecretValue(child); err != nil {
return err
}
}
case []any:
for _, child := range value {
if err := rejectSecretValue(child); err != nil {
return err
}
}
}
return nil
}

View File

@@ -0,0 +1,364 @@
package catalog
import (
"bytes"
"context"
"encoding/json"
"fmt"
"io/fs"
"os"
"path/filepath"
"reflect"
"sort"
"strings"
"testing"
"testing/fstest"
openrouter "gitea.maximumdirect.net/eric/promptkit-backend-openrouter"
rakestrawhome "gitea.maximumdirect.net/eric/promptkit-backend-rakestrawhome"
)
func TestLoadPublishedCatalogsMatchCompatibilityFixture(t *testing.T) {
loaded, err := Load(
Source{Name: "OpenRouter", ExpectedBackendID: "openrouter", FS: openrouter.FS(), Root: openrouter.Root},
Source{Name: "Rakestrawhome", ExpectedBackendID: "rakestrawhome", FS: rakestrawhome.FS(), Root: rakestrawhome.Root},
)
if err != nil {
t.Fatalf("load published catalogs: %v", err)
}
expected := loadCompatibilityFixture(t)
expectedIDs := fixtureProfileIDs(t, expected)
if !reflect.DeepEqual(loaded.profileIDs, expectedIDs) {
t.Fatalf("published profile IDs differ from compatibility fixture: got %q, want %q", loaded.profileIDs, expectedIDs)
}
actual := catalogValue(t, loaded)
actualJSON, err := json.Marshal(actual)
if err != nil {
t.Fatalf("encode loaded catalogs: %v", err)
}
expectedJSON, err := json.Marshal(expected)
if err != nil {
t.Fatalf("encode compatibility fixture: %v", err)
}
if !bytes.Equal(actualJSON, expectedJSON) {
t.Fatalf("published catalogs differ from compatibility fixture: got %#v, want %#v", actual, expected)
}
}
func TestLoadRejectsInvalidSources(t *testing.T) {
for name, sources := range map[string][]Source{
"none": nil,
"blank name": {{Name: " ", ExpectedBackendID: "openrouter", FS: openrouter.FS(), Root: openrouter.Root}},
"duplicate name": {
{Name: "same", ExpectedBackendID: "openrouter", FS: openrouter.FS(), Root: openrouter.Root},
{Name: "same", ExpectedBackendID: "rakestrawhome", FS: rakestrawhome.FS(), Root: rakestrawhome.Root},
},
"nil filesystem": {{Name: "missing", ExpectedBackendID: "openrouter", Root: openrouter.Root}},
"invalid root": {{Name: "invalid-root", ExpectedBackendID: "openrouter", FS: openrouter.FS(), Root: "."}},
"blank expected backend": {{Name: "blank-backend", ExpectedBackendID: " ", FS: openrouter.FS(), Root: openrouter.Root}},
} {
t.Run(name, func(t *testing.T) {
if _, err := Load(sources...); err == nil {
t.Fatal("expected error")
}
})
}
}
func TestLoadRejectsInvalidLayouts(t *testing.T) {
tests := map[string]func(fstest.MapFS){
"unexpected file": func(fsys fstest.MapFS) {
fsys["catalog/notes.txt"] = &fstest.MapFile{Data: []byte("unexpected")}
},
"unexpected directory": func(fsys fstest.MapFS) {
fsys["catalog/unexpected"] = &fstest.MapFile{Mode: fs.ModeDir}
},
"nonregular manifest": func(fsys fstest.MapFS) {
fsys["catalog/backend.json"].Mode = fs.ModeSymlink
},
"wrong profile extension": func(fsys fstest.MapFS) {
fsys["catalog/profiles/extra.yaml"] = &fstest.MapFile{Data: []byte(validProfile("extra", "one"))}
},
"nonregular profile": func(fsys fstest.MapFS) {
fsys["catalog/profiles/one-profile.yml"].Mode = fs.ModeSymlink
},
}
for name, mutate := range tests {
t.Run(name, func(t *testing.T) {
fsys := validCatalogFS("one")
mutate(fsys)
if _, err := Load(testSource("one", fsys)); err == nil {
t.Fatal("expected invalid layout error")
}
})
}
}
func TestLoadRejectsInvalidManifests(t *testing.T) {
tests := map[string]string{
"malformed": `{`,
"trailing value": validManifest("one", "TEST_API_KEY", "null") + `{}`,
"missing fields": `{"schema_version":1,"id":"one"}`,
"unsupported version": strings.Replace(validManifest("one", "TEST_API_KEY", "null"), `"schema_version":1`, `"schema_version":2`, 1),
"unknown field": strings.Replace(validManifest("one", "TEST_API_KEY", "null"), `"extra_params":null`, `"extra_params":null,"unknown":true`, 1),
"blank API key env": validManifest("one", " ", "null"),
"invalid API key env": validManifest("one", "LEAK-MARKER", "null"),
"invalid endpoint": strings.Replace(validManifest("one", "TEST_API_KEY", "null"), `https://one.example/v1`, `ftp://leak-marker.invalid/v1`, 1),
"zero concurrency": strings.Replace(validManifest("one", "TEST_API_KEY", "null"), `"concurrency_limit":2`, `"concurrency_limit":0`, 1),
"non-object parameters": validManifest("one", "TEST_API_KEY", `[]`),
"secret parameter": validManifest("one", "TEST_API_KEY", `{"nested":{"token":"leak-marker"}}`),
}
for name, manifest := range tests {
t.Run(name, func(t *testing.T) {
fsys := validCatalogFS("one")
fsys["catalog/backend.json"].Data = []byte(manifest)
_, err := Load(testSource("one", fsys))
if err == nil {
t.Fatal("expected invalid manifest error")
}
if strings.Contains(strings.ToLower(err.Error()), "leak-marker") {
t.Fatalf("catalog error exposed manifest content: %v", err)
}
})
}
}
func TestLoadPreservesManifestJSONNumbers(t *testing.T) {
fsys := validCatalogFS("one")
fsys["catalog/backend.json"].Data = []byte(validManifest(
"one",
"TEST_API_KEY",
`{"large":9007199254740993,"nested":[1.25]}`,
))
loaded, err := Load(testSource("one", fsys))
if err != nil {
t.Fatalf("load catalog: %v", err)
}
if got := loaded.Backends[0].ExtraParams["large"]; got != json.Number("9007199254740993") {
t.Fatalf("large JSON integer = %#v, want preserved json.Number", got)
}
nested := loaded.Backends[0].ExtraParams["nested"].([]any)
if nested[0] != json.Number("1.25") {
t.Fatalf("nested JSON number = %#v, want preserved json.Number", nested[0])
}
}
func TestLoadRejectsInvalidCatalogProfiles(t *testing.T) {
tests := map[string]func(fstest.MapFS){
"empty": func(fsys fstest.MapFS) {
delete(fsys, "catalog/profiles/one-profile.yml")
fsys["catalog/profiles"] = &fstest.MapFile{Mode: fs.ModeDir}
},
"malformed": func(fsys fstest.MapFS) {
fsys["catalog/profiles/one-profile.yml"].Data = []byte("id: [")
},
"raw API key": func(fsys fstest.MapFS) {
fsys["catalog/profiles/one-profile.yml"].Data = []byte(validProfile("one-profile", "one") + "api_key: leak-marker\n")
},
"endpoint field": func(fsys fstest.MapFS) {
fsys["catalog/profiles/one-profile.yml"].Data = []byte(validProfile("one-profile", "one") + "endpoint: ''\n")
},
"API key environment field": func(fsys fstest.MapFS) {
fsys["catalog/profiles/one-profile.yml"].Data = []byte(validProfile("one-profile", "one") + "api_key_env: ''\n")
},
"owner mismatch": func(fsys fstest.MapFS) {
fsys["catalog/profiles/one-profile.yml"].Data = []byte(validProfile("one-profile", "other"))
},
"missing base": func(fsys fstest.MapFS) {
fsys["catalog/profiles/one-profile.yml"].Data = []byte("id: one-profile\nbase_profile: leak-marker\n")
},
"cyclic base": func(fsys fstest.MapFS) {
fsys["catalog/profiles/one-profile.yml"].Data = []byte("id: one-profile\nbase_profile: second\n")
fsys["catalog/profiles/second.yml"] = &fstest.MapFile{Data: []byte("id: second\nbase_profile: one-profile\n")}
},
"secret profile parameter": func(fsys fstest.MapFS) {
fsys["catalog/profiles/one-profile.yml"].Data = []byte(validProfile("one-profile", "one") + "extra_params:\n nested:\n password: leak-marker\n")
},
"unknown field is redacted": func(fsys fstest.MapFS) {
fsys["catalog/profiles/one-profile.yml"].Data = []byte(validProfile("one-profile", "one") + "leak_marker: leak-marker\n")
},
}
for name, mutate := range tests {
t.Run(name, func(t *testing.T) {
fsys := validCatalogFS("one")
mutate(fsys)
_, err := Load(testSource("one", fsys))
if err == nil {
t.Fatal("expected invalid profile error")
}
if strings.Contains(strings.ToLower(err.Error()), "leak-marker") {
t.Fatalf("catalog error exposed profile content: %v", err)
}
})
}
}
func TestLoadRejectsCrossCatalogConflicts(t *testing.T) {
t.Run("duplicate backend", func(t *testing.T) {
second := catalogFS("one", map[string]string{
"catalog/profiles/second.yml": validProfile("second", "one"),
}, "null")
_, err := Load(
Source{Name: "first", ExpectedBackendID: "one", FS: validCatalogFS("one"), Root: "catalog"},
Source{Name: "second", ExpectedBackendID: "one", FS: second, Root: "catalog"},
)
if err == nil {
t.Fatal("expected duplicate backend error")
}
})
t.Run("duplicate profile", func(t *testing.T) {
first := catalogFS("one", map[string]string{
"catalog/profiles/shared.yml": validProfile("shared", "one"),
}, "null")
second := catalogFS("two", map[string]string{
"catalog/profiles/shared.yml": validProfile("shared", "two"),
}, "null")
_, err := Load(testSource("one", first), testSource("two", second))
if err == nil {
t.Fatal("expected duplicate profile error")
}
})
t.Run("cross-catalog base", func(t *testing.T) {
first := catalogFS("one", map[string]string{
"catalog/profiles/base.yml": validProfile("base", "one"),
}, "null")
second := catalogFS("two", map[string]string{
"catalog/profiles/child.yml": "id: child\nbase_profile: base\n",
}, "null")
_, err := Load(testSource("one", first), testSource("two", second))
if err == nil {
t.Fatal("expected cross-catalog base error")
}
})
}
func TestLoadReturnsDefensiveCatalogValues(t *testing.T) {
fsys := catalogFS("one", map[string]string{
"catalog/profiles/one-profile.yml": validProfile("one-profile", "one") + "extra_params:\n nested:\n value: profile\n",
}, `{"nested":{"value":"backend"}}`)
loaded, err := Load(testSource("one", fsys))
if err != nil {
t.Fatalf("load catalog: %v", err)
}
loaded.Backends[0].ExtraParams["nested"].(map[string]any)["value"] = "changed"
profileValue, err := loaded.Profiles.GetProfile(context.Background(), "one-profile")
if err != nil {
t.Fatalf("load profile: %v", err)
}
profileValue.ExtraParams["nested"].(map[string]any)["value"] = "changed"
again, err := Load(testSource("one", fsys))
if err != nil {
t.Fatalf("reload catalog: %v", err)
}
if got := again.Backends[0].ExtraParams["nested"].(map[string]any)["value"]; got != "backend" {
t.Fatalf("backend mutation escaped returned set: %#v", got)
}
againProfile, err := loaded.Profiles.GetProfile(context.Background(), "one-profile")
if err != nil {
t.Fatalf("reload profile: %v", err)
}
if got := againProfile.ExtraParams["nested"].(map[string]any)["value"]; got != "profile" {
t.Fatalf("profile mutation escaped returned value: %#v", got)
}
}
func loadCompatibilityFixture(t *testing.T) map[string]any {
t.Helper()
data, err := os.ReadFile(filepath.Join("..", "..", "testdata", "builtin-catalog-v1.json"))
if err != nil {
t.Fatalf("read fixture: %v", err)
}
var value map[string]any
if err := json.Unmarshal(data, &value); err != nil {
t.Fatalf("decode fixture: %v", err)
}
return value
}
func fixtureProfileIDs(t *testing.T, fixture map[string]any) []string {
t.Helper()
profiles, ok := fixture["profiles"].([]any)
if !ok {
t.Fatal("compatibility fixture profiles are malformed")
}
ids := make([]string, 0, len(profiles))
for _, entry := range profiles {
profileValue, ok := entry.(map[string]any)
if !ok {
t.Fatal("compatibility fixture profile is malformed")
}
id, ok := profileValue["id"].(string)
if !ok {
t.Fatal("compatibility fixture profile ID is malformed")
}
ids = append(ids, id)
}
sort.Strings(ids)
return ids
}
func catalogValue(t *testing.T, loaded Set) map[string]any {
t.Helper()
backends := make([]any, 0, len(loaded.Backends))
for _, backend := range loaded.Backends {
backends = append(backends, map[string]any{"id": backend.ID, "endpoint": backend.Endpoint, "api_key_env": backend.APIKeyEnv, "extra_params": interfaceValue(backend.ExtraParams), "concurrency_limit": backend.ConcurrencyLimit, "queue_capacity": backend.QueueCapacity, "queue_capacity_set": backend.QueueCapacitySet})
}
sort.Slice(backends, func(left, right int) bool {
return backends[left].(map[string]any)["id"].(string) < backends[right].(map[string]any)["id"].(string)
})
actualProfiles := make([]any, 0, len(loaded.profileIDs))
for _, id := range loaded.profileIDs {
profile, err := loaded.Profiles.GetProfile(context.Background(), id)
if err != nil {
t.Fatalf("load profile %q: %v", id, err)
}
actualProfiles = append(actualProfiles, map[string]any{"id": profile.ID, "base_profile": profile.BaseProfileID, "backend": profile.BackendID, "endpoint": profile.Endpoint, "model": profile.Model, "temperature": profile.Temperature, "max_tokens": profile.MaxTokens, "top_p": profile.TopP, "timeout_seconds": profile.TimeoutSeconds, "service_tier": profile.ServiceTier, "reasoning_effort": profile.ReasoningEffort, "api_key_env": profile.APIKeyEnv, "api_key_required": profile.APIKeyRequired, "extra_params": interfaceValue(profile.ExtraParams)})
}
return map[string]any{"backends": backends, "profiles": actualProfiles}
}
func interfaceValue(value map[string]any) any {
if value == nil {
return nil
}
return value
}
func testSource(id string, fsys fs.FS) Source {
return Source{Name: id, ExpectedBackendID: id, FS: fsys, Root: "catalog"}
}
func validCatalogFS(id string) fstest.MapFS {
return catalogFS(id, map[string]string{
"catalog/profiles/" + id + "-profile.yml": validProfile(id+"-profile", id),
}, "null")
}
func catalogFS(id string, profiles map[string]string, extraParams string) fstest.MapFS {
fsys := fstest.MapFS{
"catalog/backend.json": &fstest.MapFile{Data: []byte(validManifest(id, "TEST_API_KEY", extraParams))},
}
for name, content := range profiles {
fsys[name] = &fstest.MapFile{Data: []byte(content)}
}
return fsys
}
func validManifest(id, apiKeyEnv, extraParams string) string {
return fmt.Sprintf(
`{"schema_version":1,"id":%q,"endpoint":%q,"api_key_env":%q,"concurrency_limit":2,"queue_capacity":3,"extra_params":%s}`,
id,
"https://"+id+".example/v1",
apiKeyEnv,
extraParams,
)
}
func validProfile(id, backendID string) string {
return fmt.Sprintf("id: %s\nbackend: %s\nmodel: test-model\n", id, backendID)
}
var _ fs.FS = openrouter.FS()

View File

@@ -14,21 +14,12 @@ const (
ContentTypeApplicationJSON = "application/json" ContentTypeApplicationJSON = "application/json"
OpenAIChatCompletionsPath = "/chat/completions" OpenAIChatCompletionsPath = "/chat/completions"
ExecutionDefaultTemperature = 0.0
ExecutionDefaultMaxTokens = 0
ExecutionDefaultTopP = 1.0
ExecutionDefaultTimeoutSeconds = 600 ExecutionDefaultTimeoutSeconds = 600
) LLMRequestTimeoutDefault = 10 * time.Minute
var (
LLMRequestTimeoutDefault = 10 * time.Minute
) )
func ExecutionTargetDefault() domain.ExecutionTarget { func ExecutionTargetDefault() domain.ExecutionTarget {
return domain.ExecutionTarget{ return domain.ExecutionTarget{
Temperature: ExecutionDefaultTemperature,
MaxTokens: ExecutionDefaultMaxTokens,
TopP: ExecutionDefaultTopP,
TimeoutSeconds: ExecutionDefaultTimeoutSeconds, TimeoutSeconds: ExecutionDefaultTimeoutSeconds,
} }
} }

View File

@@ -60,15 +60,16 @@ type CacheControl struct {
// RunRequest represents a request to generate a single artifact. // RunRequest represents a request to generate a single artifact.
type RunRequest struct { type RunRequest struct {
PromptID string PromptID string
PromptVersion string PromptVersion string
ProfileID string ProfileID string
SessionID string SessionID string
APIKey string `json:"-" yaml:"-"` APIKey string `json:"-" yaml:"-"`
Inputs map[string]ArtifactRef Inputs map[string]ArtifactRef
Vars map[string]string Vars map[string]string
Execution *ExecutionTargetOverride Execution *ExecutionTargetOverride
Validation *OutputContract Validation *OutputContract
AppendedMessages []RenderedMessage
} }
// RunResult represents the complete result of a prompt execution run. // RunResult represents the complete result of a prompt execution run.
@@ -97,22 +98,22 @@ type RunResult struct {
// PreparedRun contains pre-LLM execution state from the prepare/render phase. // PreparedRun contains pre-LLM execution state from the prepare/render phase.
// It must never include resolved API key values, model output, or validation data. // It must never include resolved API key values, model output, or validation data.
type PreparedRun struct { type PreparedRun struct {
PromptID string `json:"prompt_id"` PromptID string
PromptVersion string `json:"prompt_version,omitempty"` PromptVersion string
PromptHash string `json:"prompt_hash,omitempty"` PromptHash string
SelectedProfileID string `json:"selected_profile_id"` SelectedProfileID string
SelectedBackendID string `json:"selected_backend_id,omitempty"` SelectedBackendID string
EffectiveModelParams ExecutionTarget `json:"effective_model_params"` EffectiveModelParams ExecutionTarget
TargetPresence ExecutionTargetPresence `json:"-"` TargetPresence ExecutionTargetPresence
OutputContract OutputContract `json:"output_contract"` OutputContract OutputContract
StructuredOutput *StructuredOutputSpec `json:"structured_output,omitempty"` StructuredOutput *StructuredOutputSpec
InputHashes map[string]string `json:"input_hashes,omitempty"` InputHashes map[string]string
SessionID string `json:"session_id,omitempty"` SessionID string
RenderedPromptHash string `json:"rendered_prompt_hash"` RenderedPromptHash string
Messages []RenderedMessage `json:"messages"` Messages []RenderedMessage
StartTime time.Time `json:"start_time,omitempty"` StartTime time.Time
EndTime time.Time `json:"end_time,omitempty"` EndTime time.Time
DurationMS int64 `json:"duration_ms,omitempty"` DurationMS int64
} }
// ArtifactRef represents a reference to an input artifact. // ArtifactRef represents a reference to an input artifact.
@@ -145,6 +146,16 @@ type PromptDefinition struct {
Validation OutputContract `yaml:"validation"` Validation OutputContract `yaml:"validation"`
} }
// PromptInspection is the resolved result of exact prompt inspection.
type PromptInspection struct {
PromptID string
PromptVersion string
PromptHash string
DefaultProfileID string
Inputs []PromptInput
OutputContract OutputContract
}
// PromptInput describes one named input expected by a prompt definition. // PromptInput describes one named input expected by a prompt definition.
type PromptInput struct { type PromptInput struct {
Name string `yaml:"name"` Name string `yaml:"name"`
@@ -182,6 +193,7 @@ type BackendCapacityPolicy struct {
// ExecutionProfile describes how and where to execute a model. // ExecutionProfile describes how and where to execute a model.
type ExecutionProfile struct { type ExecutionProfile struct {
ID string `yaml:"id"` ID string `yaml:"id"`
BaseProfileID string `yaml:"base_profile"`
BackendID string `yaml:"backend"` BackendID string `yaml:"backend"`
Endpoint string `yaml:"endpoint"` Endpoint string `yaml:"endpoint"`
Model string `yaml:"model"` Model string `yaml:"model"`
@@ -236,6 +248,13 @@ type ExecutionTarget struct {
ExtraParams map[string]any `yaml:"extra_params" json:"extra_params"` ExtraParams map[string]any `yaml:"extra_params" json:"extra_params"`
} }
// ProfileInspection is the resolved result of exact profile inspection.
type ProfileInspection struct {
ProfileID string
EffectiveModelParams ExecutionTarget
APIKeyRequired bool
}
// OutputContract defines the requirements for the output artifact. // OutputContract defines the requirements for the output artifact.
type OutputContract struct { type OutputContract struct {
Format OutputFormat `yaml:"format"` Format OutputFormat `yaml:"format"`

View File

@@ -0,0 +1,39 @@
package domain
import (
"errors"
"net/url"
"strings"
)
// NormalizeOpenAICompatibleBaseEndpoint trims and validates a source-neutral
// OpenAI-compatible provider base endpoint.
func NormalizeOpenAICompatibleBaseEndpoint(endpoint string) (string, error) {
endpoint = strings.TrimSpace(endpoint)
if endpoint == "" {
return "", errors.New("endpoint must not be blank")
}
if strings.Contains(endpoint, "#") {
return "", errors.New("endpoint must not contain a fragment")
}
parsed, err := url.Parse(endpoint)
if err != nil {
return "", errors.New("endpoint must be a valid URL")
}
parsed.Scheme = strings.ToLower(parsed.Scheme)
if parsed.Scheme != "http" && parsed.Scheme != "https" {
return "", errors.New("endpoint must use http or https")
}
if !parsed.IsAbs() || parsed.Hostname() == "" {
return "", errors.New("endpoint must be absolute and include a host")
}
if parsed.User != nil {
return "", errors.New("endpoint must not contain user information")
}
if parsed.RawQuery != "" || parsed.ForceQuery {
return "", errors.New("endpoint must not contain a query string")
}
return parsed.String(), nil
}

View File

@@ -0,0 +1,47 @@
package domain
import "testing"
func TestNormalizeOpenAICompatibleBaseEndpoint(t *testing.T) {
tests := []struct {
name string
endpoint string
want string
wantErr bool
}{
{name: "http host", endpoint: "http://provider.example", want: "http://provider.example"},
{name: "https nested path and whitespace", endpoint: " HTTPS://provider.example/api/openai/v1 ", want: "https://provider.example/api/openai/v1"},
{name: "IPv4 host and port", endpoint: "http://127.0.0.1:8080/v1", want: "http://127.0.0.1:8080/v1"},
{name: "IPv6 host and port", endpoint: "https://[::1]:8443/v1", want: "https://[::1]:8443/v1"},
{name: "repeated trailing slashes", endpoint: "https://provider.example/v1///", want: "https://provider.example/v1///"},
{name: "blank", endpoint: " \t\n ", wantErr: true},
{name: "relative path", endpoint: "/api/v1", wantErr: true},
{name: "scheme relative", endpoint: "//provider.example/v1", wantErr: true},
{name: "missing host", endpoint: "https:///v1", wantErr: true},
{name: "unsupported scheme", endpoint: "ftp://provider.example/v1", wantErr: true},
{name: "user information", endpoint: "https://user:secret@provider.example/v1", wantErr: true},
{name: "query", endpoint: "https://provider.example/v1?mode=chat", wantErr: true},
{name: "empty query", endpoint: "https://provider.example/v1?", wantErr: true},
{name: "fragment", endpoint: "https://provider.example/v1#chat", wantErr: true},
{name: "empty fragment", endpoint: "https://provider.example/v1#", wantErr: true},
{name: "malformed URL", endpoint: "https://provider.example/%zz", wantErr: true},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
got, err := NormalizeOpenAICompatibleBaseEndpoint(tc.endpoint)
if tc.wantErr {
if err == nil {
t.Fatalf("expected endpoint error, got %q", got)
}
return
}
if err != nil {
t.Fatalf("normalize endpoint: %v", err)
}
if got != tc.want {
t.Fatalf("normalized endpoint = %q, want %q", got, tc.want)
}
})
}
}

View File

@@ -0,0 +1,34 @@
package domain
import (
"errors"
"math"
"time"
)
const maxExecutionTimeoutSeconds int64 = math.MaxInt64 / int64(time.Second)
// ValidateExecutionTargetSettings validates source-neutral execution-setting
// invariants on a resolved target.
func ValidateExecutionTargetSettings(target ExecutionTarget) error {
if !isFinite(target.Temperature) || target.Temperature < 0 || target.Temperature > 2 {
return errors.New("temperature must be finite and between 0 and 2")
}
if target.MaxTokens < 0 {
return errors.New("max_tokens must be greater than or equal to 0")
}
if !isFinite(target.TopP) || target.TopP < 0 || target.TopP > 1 {
return errors.New("top_p must be finite and between 0 and 1")
}
if target.TimeoutSeconds < 0 {
return errors.New("timeout_seconds must be greater than or equal to 0")
}
if int64(target.TimeoutSeconds) > maxExecutionTimeoutSeconds {
return errors.New("timeout_seconds exceeds the maximum supported duration")
}
return nil
}
func isFinite(value float64) bool {
return !math.IsNaN(value) && !math.IsInf(value, 0)
}

View File

@@ -0,0 +1,73 @@
package domain
import (
"math"
"strconv"
"strings"
"testing"
)
func TestValidateExecutionTargetSettings(t *testing.T) {
valid := ExecutionTarget{
Temperature: 1,
MaxTokens: 1,
TopP: 0.5,
TimeoutSeconds: 1,
}
type testCase struct {
name string
change func(*ExecutionTarget)
wantErr string
}
tests := []testCase{
{name: "temperature lower boundary", change: func(v *ExecutionTarget) { v.Temperature = 0 }},
{name: "temperature finite lower neighbor", change: func(v *ExecutionTarget) { v.Temperature = math.Nextafter(0, 1) }},
{name: "temperature finite upper neighbor", change: func(v *ExecutionTarget) { v.Temperature = math.Nextafter(2, 0) }},
{name: "temperature upper boundary", change: func(v *ExecutionTarget) { v.Temperature = 2 }},
{name: "temperature below lower boundary", change: func(v *ExecutionTarget) { v.Temperature = math.Nextafter(0, math.Inf(-1)) }, wantErr: "temperature"},
{name: "temperature above upper boundary", change: func(v *ExecutionTarget) { v.Temperature = math.Nextafter(2, math.Inf(1)) }, wantErr: "temperature"},
{name: "temperature NaN", change: func(v *ExecutionTarget) { v.Temperature = math.NaN() }, wantErr: "temperature"},
{name: "temperature positive infinity", change: func(v *ExecutionTarget) { v.Temperature = math.Inf(1) }, wantErr: "temperature"},
{name: "temperature negative infinity", change: func(v *ExecutionTarget) { v.Temperature = math.Inf(-1) }, wantErr: "temperature"},
{name: "max tokens lower boundary", change: func(v *ExecutionTarget) { v.MaxTokens = 0 }},
{name: "max tokens finite neighbor", change: func(v *ExecutionTarget) { v.MaxTokens = 1 }},
{name: "max tokens below lower boundary", change: func(v *ExecutionTarget) { v.MaxTokens = -1 }, wantErr: "max_tokens"},
{name: "top p lower boundary", change: func(v *ExecutionTarget) { v.TopP = 0 }},
{name: "top p finite lower neighbor", change: func(v *ExecutionTarget) { v.TopP = math.Nextafter(0, 1) }},
{name: "top p finite upper neighbor", change: func(v *ExecutionTarget) { v.TopP = math.Nextafter(1, 0) }},
{name: "top p upper boundary", change: func(v *ExecutionTarget) { v.TopP = 1 }},
{name: "top p below lower boundary", change: func(v *ExecutionTarget) { v.TopP = math.Nextafter(0, math.Inf(-1)) }, wantErr: "top_p"},
{name: "top p above upper boundary", change: func(v *ExecutionTarget) { v.TopP = math.Nextafter(1, math.Inf(1)) }, wantErr: "top_p"},
{name: "top p NaN", change: func(v *ExecutionTarget) { v.TopP = math.NaN() }, wantErr: "top_p"},
{name: "top p positive infinity", change: func(v *ExecutionTarget) { v.TopP = math.Inf(1) }, wantErr: "top_p"},
{name: "top p negative infinity", change: func(v *ExecutionTarget) { v.TopP = math.Inf(-1) }, wantErr: "top_p"},
{name: "timeout lower boundary", change: func(v *ExecutionTarget) { v.TimeoutSeconds = 0 }},
{name: "timeout finite neighbor", change: func(v *ExecutionTarget) { v.TimeoutSeconds = 1 }},
{name: "timeout below lower boundary", change: func(v *ExecutionTarget) { v.TimeoutSeconds = -1 }, wantErr: "timeout_seconds"},
}
if strconv.IntSize == 64 {
durationLimit := maxExecutionTimeoutSeconds
tests = append(tests,
testCase{name: "timeout duration boundary", change: func(v *ExecutionTarget) { v.TimeoutSeconds = int(durationLimit) }},
testCase{name: "timeout above duration boundary", change: func(v *ExecutionTarget) { v.TimeoutSeconds = int(durationLimit) + 1 }, wantErr: "timeout_seconds"},
)
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
target := valid
tt.change(&target)
err := ValidateExecutionTargetSettings(target)
if tt.wantErr == "" {
if err != nil {
t.Fatalf("validate execution settings: %v", err)
}
return
}
if err == nil || !strings.Contains(err.Error(), tt.wantErr) {
t.Fatalf("error = %v, want diagnostic containing %q", err, tt.wantErr)
}
})
}
}

View File

@@ -0,0 +1,82 @@
package domain
import (
"errors"
"strings"
"unicode/utf8"
)
const (
RoleDeveloper = "developer"
RoleSystem = "system"
RoleUser = "user"
RoleAssistant = "assistant"
)
// NormalizeMessageRole validates and canonicalizes a provider-bound chat role.
func NormalizeMessageRole(role string) (string, error) {
if !utf8.ValidString(role) {
return "", errors.New("message role must be valid UTF-8")
}
normalized := strings.ToLower(strings.TrimSpace(role))
switch normalized {
case RoleDeveloper, RoleSystem, RoleUser, RoleAssistant:
return normalized, nil
default:
return "", errors.New("message role must be developer, system, user, or assistant")
}
}
// NormalizeCacheControl validates, canonicalizes, and copies cache metadata.
func NormalizeCacheControl(control *CacheControl) (*CacheControl, error) {
if control == nil {
return nil, nil
}
if !utf8.ValidString(string(control.Type)) {
return nil, errors.New("cache control type must be valid UTF-8")
}
if !utf8.ValidString(control.TTL) {
return nil, errors.New("cache control ttl must be valid UTF-8")
}
cacheType := strings.TrimSpace(string(control.Type))
if cacheType == "" {
return nil, errors.New("cache control type is required")
}
if CacheControlType(cacheType) != CacheControlEphemeral {
return nil, errors.New("unsupported type")
}
ttl := strings.TrimSpace(control.TTL)
if ttl != "" && ttl != "1h" {
return nil, errors.New("unsupported ttl")
}
return &CacheControl{Type: CacheControlType(cacheType), TTL: ttl}, nil
}
// CloneRenderedMessages returns a deep copy of rendered messages.
func CloneRenderedMessages(messages []RenderedMessage) []RenderedMessage {
cloned := make([]RenderedMessage, len(messages))
copyRenderedMessages(cloned, messages)
return cloned
}
// ConcatRenderedMessages returns an independently owned concatenation of messages.
func ConcatRenderedMessages(prefix, suffix []RenderedMessage) []RenderedMessage {
messages := make([]RenderedMessage, len(prefix)+len(suffix))
copyRenderedMessages(messages, prefix)
copyRenderedMessages(messages[len(prefix):], suffix)
return messages
}
func copyRenderedMessages(destination, source []RenderedMessage) {
for index, message := range source {
destination[index] = message
if message.CacheControl != nil {
cacheControl := *message.CacheControl
destination[index].CacheControl = &cacheControl
}
}
}

View File

@@ -0,0 +1,121 @@
package domain
import (
"strings"
"testing"
)
func TestNormalizeMessageRole(t *testing.T) {
tests := []struct {
name string
input string
want string
wantErr bool
}{
{name: "developer", input: RoleDeveloper, want: RoleDeveloper},
{name: "system", input: RoleSystem, want: RoleSystem},
{name: "user", input: RoleUser, want: RoleUser},
{name: "assistant", input: RoleAssistant, want: RoleAssistant},
{name: "surrounding whitespace and mixed case", input: " \u2003UsEr\u2003 ", want: RoleUser},
{name: "blank", input: " \t\n ", wantErr: true},
{name: "tool", input: "tool", wantErr: true},
{name: "function", input: "function", wantErr: true},
{name: "custom", input: "custom-role", wantErr: true},
{name: "invalid UTF-8", input: string([]byte{0xff}), wantErr: true},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
got, err := NormalizeMessageRole(test.input)
if test.wantErr {
if err == nil {
t.Fatal("expected an error")
}
if strings.Contains(err.Error(), test.input) {
t.Fatalf("error exposed the input: %v", err)
}
return
}
if err != nil {
t.Fatalf("NormalizeMessageRole() error = %v", err)
}
if got != test.want {
t.Fatalf("NormalizeMessageRole() = %q, want %q", got, test.want)
}
})
}
}
func TestNormalizeCacheControl(t *testing.T) {
tests := []struct {
name string
input *CacheControl
want *CacheControl
wantErr bool
}{
{name: "nil", input: nil, want: nil},
{
name: "canonical values",
input: &CacheControl{Type: CacheControlEphemeral, TTL: "1h"},
want: &CacheControl{Type: CacheControlEphemeral, TTL: "1h"},
},
{
name: "trims values",
input: &CacheControl{Type: " ephemeral ", TTL: " 1h\t"},
want: &CacheControl{Type: CacheControlEphemeral, TTL: "1h"},
},
{name: "empty type", input: &CacheControl{}, wantErr: true},
{name: "unsupported type", input: &CacheControl{Type: "persistent"}, wantErr: true},
{name: "unsupported ttl", input: &CacheControl{Type: CacheControlEphemeral, TTL: "5m"}, wantErr: true},
{name: "invalid type UTF-8", input: &CacheControl{Type: CacheControlType(string([]byte{0xff}))}, wantErr: true},
{name: "invalid ttl UTF-8", input: &CacheControl{Type: CacheControlEphemeral, TTL: string([]byte{0xff})}, wantErr: true},
}
for _, test := range tests {
t.Run(test.name, func(t *testing.T) {
got, err := NormalizeCacheControl(test.input)
if test.wantErr {
if err == nil {
t.Fatal("expected an error")
}
return
}
if err != nil {
t.Fatalf("NormalizeCacheControl() error = %v", err)
}
if got == nil || test.want == nil {
if got != test.want {
t.Fatalf("NormalizeCacheControl() = %#v, want %#v", got, test.want)
}
return
}
if *got != *test.want {
t.Fatalf("NormalizeCacheControl() = %#v, want %#v", got, test.want)
}
if got == test.input {
t.Fatal("normalized cache control aliases its input")
}
})
}
}
func TestRenderedMessageCloning(t *testing.T) {
prefix := []RenderedMessage{{Role: RoleSystem, Content: "prefix", CacheControl: &CacheControl{Type: CacheControlEphemeral, TTL: "1h"}}}
suffix := []RenderedMessage{{Role: RoleUser, Content: " suffix "}, {Role: RoleAssistant, Content: "", CacheControl: &CacheControl{Type: CacheControlEphemeral}}}
cloned := CloneRenderedMessages(prefix)
combined := ConcatRenderedMessages(prefix, suffix)
if len(combined) != 3 || combined[0].Content != "prefix" || combined[1].Content != " suffix " || combined[2].Content != "" {
t.Fatalf("unexpected combined messages: %#v", combined)
}
if cloned[0].CacheControl == prefix[0].CacheControl || combined[0].CacheControl == prefix[0].CacheControl || combined[2].CacheControl == suffix[1].CacheControl {
t.Fatal("cloned cache controls alias their inputs")
}
prefix[0].Content = "changed"
prefix[0].CacheControl.TTL = ""
suffix[1].CacheControl.Type = "changed"
if cloned[0].Content != "prefix" || cloned[0].CacheControl.TTL != "1h" || combined[0].Content != "prefix" || combined[0].CacheControl.TTL != "1h" || combined[2].CacheControl.Type != CacheControlEphemeral {
t.Fatalf("cloned messages changed with their inputs: cloned=%#v combined=%#v", cloned, combined)
}
}

View File

@@ -0,0 +1,38 @@
package domain
import (
"errors"
"fmt"
"strings"
)
const maxOutputRepairAttempts = 3
// ValidateOutputContract validates source-neutral output-contract invariants.
func ValidateOutputContract(contract OutputContract) error {
switch contract.Format {
case FormatText, FormatMarkdown, FormatJSON:
default:
return fmt.Errorf("invalid output format: %q", contract.Format)
}
switch contract.ValidationMode {
case ValidationNone, ValidationBasic, ValidationJSON, ValidationJSONSchema:
default:
return fmt.Errorf("invalid validation mode: %q", contract.ValidationMode)
}
if contract.ValidationMode == ValidationJSONSchema && strings.TrimSpace(contract.SchemaPath) == "" {
return errors.New("schema_path is required when validation_mode is json_schema")
}
if contract.RepairAttempts < 0 {
return errors.New("repair_attempts must be greater than or equal to 0")
}
if contract.RepairAttempts > maxOutputRepairAttempts {
return fmt.Errorf("repair_attempts must be less than or equal to %d", maxOutputRepairAttempts)
}
if contract.ValidationMode == ValidationNone && contract.RepairAttempts > 0 {
return errors.New("repair_attempts requires basic, json, or json_schema validation")
}
return nil
}

View File

@@ -0,0 +1,87 @@
package domain
import (
"strings"
"testing"
)
func TestValidateOutputContract(t *testing.T) {
valid := OutputContract{
Format: FormatText,
ValidationMode: ValidationNone,
}
tests := []struct {
name string
change func(*OutputContract)
wantErr string
}{
{name: "text format", change: func(c *OutputContract) { c.Format = FormatText }},
{name: "markdown format", change: func(c *OutputContract) { c.Format = FormatMarkdown }},
{name: "json format", change: func(c *OutputContract) { c.Format = FormatJSON }},
{name: "empty format", change: func(c *OutputContract) { c.Format = "" }, wantErr: "format"},
{name: "unsupported format", change: func(c *OutputContract) { c.Format = OutputFormat("binary") }, wantErr: "format"},
{name: "none validation", change: func(c *OutputContract) { c.ValidationMode = ValidationNone }},
{name: "basic validation", change: func(c *OutputContract) { c.ValidationMode = ValidationBasic }},
{name: "json validation", change: func(c *OutputContract) { c.ValidationMode = ValidationJSON }},
{name: "json schema validation", change: func(c *OutputContract) {
c.ValidationMode = ValidationJSONSchema
c.SchemaPath = "schema.json"
}},
{name: "empty validation mode", change: func(c *OutputContract) { c.ValidationMode = "" }, wantErr: "validation mode"},
{name: "unsupported validation mode", change: func(c *OutputContract) { c.ValidationMode = ValidationMode("unknown") }, wantErr: "validation mode"},
{name: "negative repair attempts", change: func(c *OutputContract) {
c.ValidationMode = ValidationBasic
c.RepairAttempts = -1
}, wantErr: "repair_attempts"},
{name: "zero repair attempts", change: func(c *OutputContract) { c.RepairAttempts = 0 }},
{name: "one repair attempt", change: func(c *OutputContract) {
c.ValidationMode = ValidationBasic
c.RepairAttempts = 1
}},
{name: "maximum repair attempts", change: func(c *OutputContract) {
c.ValidationMode = ValidationJSON
c.RepairAttempts = 3
}},
{name: "too many repair attempts", change: func(c *OutputContract) {
c.ValidationMode = ValidationJSONSchema
c.SchemaPath = "schema.json"
c.RepairAttempts = 4
}, wantErr: "repair_attempts"},
{name: "none validation with repair attempts", change: func(c *OutputContract) {
c.ValidationMode = ValidationNone
c.RepairAttempts = 1
}, wantErr: "repair_attempts"},
{name: "json schema empty path", change: func(c *OutputContract) {
c.ValidationMode = ValidationJSONSchema
c.SchemaPath = ""
}, wantErr: "schema_path"},
{name: "json schema whitespace path", change: func(c *OutputContract) {
c.ValidationMode = ValidationJSONSchema
c.SchemaPath = " \t "
}, wantErr: "schema_path"},
{name: "json schema nonblank path", change: func(c *OutputContract) {
c.ValidationMode = ValidationJSONSchema
c.SchemaPath = " schema.json "
}},
{name: "non-schema empty path", change: func(c *OutputContract) { c.SchemaPath = "" }},
{name: "non-schema populated path", change: func(c *OutputContract) { c.SchemaPath = "ignored.json" }},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
contract := valid
tt.change(&contract)
err := ValidateOutputContract(contract)
if tt.wantErr == "" {
if err != nil {
t.Fatalf("validate output contract: %v", err)
}
return
}
if err == nil || !strings.Contains(err.Error(), tt.wantErr) {
t.Fatalf("error = %v, want diagnostic containing %q", err, tt.wantErr)
}
})
}
}

View File

@@ -1,141 +0,0 @@
package domain
import (
"encoding/json"
"strings"
"testing"
)
func TestPreparedRunJSONDoesNotIncludeSecretValues(t *testing.T) {
const envName = "PROMPTKIT_TEST_API_KEY"
const secret = "super-secret-value"
t.Setenv(envName, secret)
prepared := PreparedRun{
PromptID: "prompt.id",
PromptVersion: "v1",
PromptHash: "prompt-hash",
SelectedProfileID: "local-fast",
EffectiveModelParams: ExecutionTarget{
Endpoint: "http://llm/v1",
Model: "gpt-test",
APIKeyEnv: envName,
APIKey: secret,
},
InputHashes: map[string]string{"transcript": "hash-1"},
RenderedPromptHash: "rendered-hash",
Messages: []RenderedMessage{
{Role: "system", Content: "You are helpful."},
{Role: "user", Content: "Summarize this."},
},
}
b, err := json.Marshal(prepared)
if err != nil {
t.Fatalf("marshal failed: %v", err)
}
out := string(b)
if strings.Contains(out, secret) {
t.Fatalf("prepared run JSON unexpectedly contains secret value: %s", out)
}
if !strings.Contains(out, `"api_key_env":"`+envName+`"`) {
t.Fatalf("prepared run JSON should include api_key_env name: %s", out)
}
var top map[string]any
if err := json.Unmarshal(b, &top); err != nil {
t.Fatalf("unmarshal failed: %v", err)
}
for _, forbidden := range []string{"raw_output", "validation", "artifact"} {
if _, ok := top[forbidden]; ok {
t.Fatalf("prepared run JSON should not include %q", forbidden)
}
}
}
func TestPreparedRunJSONIncludesMessageCacheControlOnlyWhenPresent(t *testing.T) {
prepared := PreparedRun{
PromptID: "prompt.id",
SelectedProfileID: "local-fast",
EffectiveModelParams: ExecutionTarget{
Endpoint: "http://llm/v1",
Model: "gpt-test",
},
RenderedPromptHash: "rendered-hash",
Messages: []RenderedMessage{
{
Role: "system",
Content: "You are helpful.",
CacheControl: &CacheControl{
Type: CacheControlEphemeral,
TTL: "1h",
},
},
{Role: "user", Content: "Summarize this."},
},
}
b, err := json.Marshal(prepared)
if err != nil {
t.Fatalf("marshal failed: %v", err)
}
var decoded struct {
Messages []map[string]any `json:"messages"`
}
if err := json.Unmarshal(b, &decoded); err != nil {
t.Fatalf("unmarshal failed: %v", err)
}
if len(decoded.Messages) != 2 {
t.Fatalf("expected 2 messages, got %d", len(decoded.Messages))
}
cacheControl, ok := decoded.Messages[0]["cache_control"].(map[string]any)
if !ok {
t.Fatalf("expected cache_control on first message, got %#v", decoded.Messages[0])
}
if cacheControl["type"] != string(CacheControlEphemeral) || cacheControl["ttl"] != "1h" {
t.Fatalf("unexpected cache_control payload: %#v", cacheControl)
}
if _, ok := decoded.Messages[1]["cache_control"]; ok {
t.Fatalf("expected second message to omit cache_control, got %#v", decoded.Messages[1])
}
}
func TestPreparedRunJSONIncludesSessionIDOnlyWhenPresent(t *testing.T) {
prepared := PreparedRun{
PromptID: "prompt.id",
SelectedProfileID: "local-fast",
EffectiveModelParams: ExecutionTarget{
Endpoint: "http://llm/v1",
Model: "gpt-test",
},
SessionID: "session-123",
RenderedPromptHash: "rendered-hash",
Messages: []RenderedMessage{{Role: "user", Content: "Summarize this."}},
}
b, err := json.Marshal(prepared)
if err != nil {
t.Fatalf("marshal failed: %v", err)
}
var decoded map[string]any
if err := json.Unmarshal(b, &decoded); err != nil {
t.Fatalf("unmarshal failed: %v", err)
}
if decoded["session_id"] != "session-123" {
t.Fatalf("expected session_id in prepared run JSON, got %#v", decoded["session_id"])
}
prepared.SessionID = ""
b, err = json.Marshal(prepared)
if err != nil {
t.Fatalf("marshal failed: %v", err)
}
if strings.Contains(string(b), "session_id") {
t.Fatalf("expected empty session_id to be omitted, got %s", b)
}
}

View File

@@ -8,6 +8,9 @@ import (
// NormalizeSessionID applies the shared session identifier rule. // NormalizeSessionID applies the shared session identifier rule.
func NormalizeSessionID(raw string) (string, error) { func NormalizeSessionID(raw string) (string, error) {
if !utf8.ValidString(raw) {
return "", fmt.Errorf("session_id must contain valid UTF-8")
}
normalized := strings.TrimSpace(raw) normalized := strings.TrimSpace(raw)
if normalized == "" { if normalized == "" {
return "", nil return "", nil

View File

@@ -7,10 +7,10 @@ import (
func TestNormalizeSessionID(t *testing.T) { func TestNormalizeSessionID(t *testing.T) {
tests := []struct { tests := []struct {
name string name string
raw string raw string
want string want string
wantErr bool wantErrContains string
}{ }{
{ {
name: "trims surrounding Unicode whitespace", name: "trims surrounding Unicode whitespace",
@@ -28,21 +28,24 @@ func TestNormalizeSessionID(t *testing.T) {
want: strings.Repeat("界", SessionIDMaxLength), want: strings.Repeat("界", SessionIDMaxLength),
}, },
{ {
name: "one Unicode code point over maximum is rejected", name: "one Unicode code point over maximum is rejected",
raw: strings.Repeat("界", SessionIDMaxLength+1), raw: strings.Repeat("界", SessionIDMaxLength+1),
wantErr: true, wantErrContains: "exceeds maximum",
}, },
{name: "invalid UTF-8 before valid content", raw: string([]byte{0xff}) + "session", wantErrContains: "valid UTF-8"},
{name: "invalid UTF-8 within valid content", raw: "ses" + string([]byte{0xff}) + "sion", wantErrContains: "valid UTF-8"},
{name: "invalid UTF-8 after valid content", raw: "session" + string([]byte{0xff}), wantErrContains: "valid UTF-8"},
} }
for _, tt := range tests { for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) { t.Run(tt.name, func(t *testing.T) {
got, err := NormalizeSessionID(tt.raw) got, err := NormalizeSessionID(tt.raw)
if tt.wantErr { if tt.wantErrContains != "" {
if err == nil { if err == nil {
t.Fatal("expected normalization error") t.Fatal("expected normalization error")
} }
if !strings.Contains(err.Error(), "exceeds maximum") { if !strings.Contains(err.Error(), tt.wantErrContains) {
t.Fatalf("expected useful length diagnostic, got %v", err) t.Fatalf("expected diagnostic containing %q, got %v", tt.wantErrContains, err)
} }
return return
} }

View File

@@ -36,7 +36,8 @@ func FindYAMLFiles(ctx context.Context, root string) ([]string, error) {
return files, err return files, err
} }
// FindFSYAMLFiles returns sorted paths for .yaml and .yml files under root in fsys. // FindFSYAMLFiles returns root itself when it names a file. For a directory
// root, it returns sorted paths for .yaml and .yml files beneath that root.
func FindFSYAMLFiles(ctx context.Context, fsys fs.FS, root string) ([]string, error) { func FindFSYAMLFiles(ctx context.Context, fsys fs.FS, root string) ([]string, error) {
cleanRoot := CleanFSRoot(root) cleanRoot := CleanFSRoot(root)
var files []string var files []string
@@ -52,6 +53,10 @@ func FindFSYAMLFiles(ctx context.Context, fsys fs.FS, root string) ([]string, er
if d.IsDir() { if d.IsDir() {
return nil return nil
} }
if name == cleanRoot {
files = append(files, name)
return nil
}
if !IsYAMLFile(d.Name()) { if !IsYAMLFile(d.Name()) {
return nil return nil
} }
@@ -71,10 +76,10 @@ func RelativePath(root string, filePath string) string {
return filepath.Clean(rel) return filepath.Clean(rel)
} }
// CleanFSRoot normalizes a root path for use with fs.FS. // CleanFSRoot normalizes a root path for use with fs.FS while preserving
// nonblank leading and trailing whitespace.
func CleanFSRoot(root string) string { func CleanFSRoot(root string) string {
root = strings.TrimSpace(root) if strings.TrimSpace(root) == "" || root == "." {
if root == "" || root == "." {
return "." return "."
} }
return path.Clean(root) return path.Clean(root)
@@ -97,22 +102,21 @@ func DisplayPath(root string, name string) string {
// ResolveFSPath resolves userPath from baseDir and keeps it inside root. // ResolveFSPath resolves userPath from baseDir and keeps it inside root.
func ResolveFSPath(root string, baseDir string, userPath string) (string, string, error) { func ResolveFSPath(root string, baseDir string, userPath string) (string, string, error) {
cleanRoot := CleanFSRoot(root) cleanRoot := CleanFSRoot(root)
cleanBase := path.Clean(strings.TrimSpace(baseDir)) cleanBase := path.Clean(baseDir)
if cleanBase == "" { if strings.TrimSpace(baseDir) == "" {
cleanBase = cleanRoot cleanBase = cleanRoot
} }
if !containsFSPath(cleanRoot, cleanBase) { if !containsFSPath(cleanRoot, cleanBase) {
return "", "", fmt.Errorf("base path %q is outside source root %q", cleanBase, cleanRoot) return "", "", fmt.Errorf("base path %q is outside source root %q", cleanBase, cleanRoot)
} }
cleanUserPath := strings.TrimSpace(userPath) if strings.TrimSpace(userPath) == "" {
if cleanUserPath == "" {
return "", "", fmt.Errorf("path is required") return "", "", fmt.Errorf("path is required")
} }
cleanUserPath = path.Clean(cleanUserPath) if path.IsAbs(userPath) {
if path.IsAbs(cleanUserPath) {
return "", "", fmt.Errorf("path %q must be relative", userPath) return "", "", fmt.Errorf("path %q must be relative", userPath)
} }
cleanUserPath := path.Clean(userPath)
resolved := path.Clean(path.Join(cleanBase, cleanUserPath)) resolved := path.Clean(path.Join(cleanBase, cleanUserPath))
if !containsFSPath(cleanRoot, resolved) { if !containsFSPath(cleanRoot, resolved) {
@@ -130,13 +134,6 @@ func containsFSPath(root string, name string) bool {
return name == root || strings.HasPrefix(name, strings.TrimSuffix(root, "/")+"/") return name == root || strings.HasPrefix(name, strings.TrimSuffix(root, "/")+"/")
} }
// Stem strips .yaml or .yml from a file name.
func Stem(name string) string {
name = strings.TrimSuffix(name, ".yaml")
name = strings.TrimSuffix(name, ".yml")
return name
}
func IsYAMLFile(name string) bool { func IsYAMLFile(name string) bool {
return strings.HasSuffix(name, ".yaml") || strings.HasSuffix(name, ".yml") return strings.HasSuffix(name, ".yaml") || strings.HasSuffix(name, ".yml")
} }

View File

@@ -54,7 +54,7 @@ func TestFindFSYAMLFilesNestedSortedAndFiltered(t *testing.T) {
"other/ignored.yaml": &fstest.MapFile{Data: []byte("id: ignored")}, "other/ignored.yaml": &fstest.MapFile{Data: []byte("id: ignored")},
} }
got, err := FindFSYAMLFiles(context.Background(), fsys, " prompts ") got, err := FindFSYAMLFiles(context.Background(), fsys, "prompts")
if err != nil { if err != nil {
t.Fatalf("expected no error, got %v", err) t.Fatalf("expected no error, got %v", err)
} }
@@ -98,8 +98,11 @@ func TestCleanFSRoot(t *testing.T) {
want string want string
}{ }{
{name: "empty", root: "", want: "."}, {name: "empty", root: "", want: "."},
{name: "whitespace only", root: " \t ", want: "."},
{name: "dot", root: ".", want: "."}, {name: "dot", root: ".", want: "."},
{name: "trimmed", root: " prompts/../profiles ", want: "profiles"}, {name: "cleaned", root: "prompts/../profiles", want: "profiles"},
{name: "leading whitespace preserved", root: " profiles", want: " profiles"},
{name: "trailing whitespace preserved", root: "profiles ", want: "profiles "},
} }
for _, tc := range tests { for _, tc := range tests {
@@ -158,6 +161,22 @@ func TestResolveFSPath(t *testing.T) {
wantPath: "prompts/shared/user.tmpl", wantPath: "prompts/shared/user.tmpl",
wantDisplay: "shared/user.tmpl", wantDisplay: "shared/user.tmpl",
}, },
{
name: "leading whitespace preserved",
root: "prompts",
baseDir: "prompts/nested",
userPath: " user.tmpl",
wantPath: "prompts/nested/ user.tmpl",
wantDisplay: "nested/ user.tmpl",
},
{
name: "trailing whitespace preserved",
root: "prompts",
baseDir: "prompts/nested",
userPath: "user.tmpl ",
wantPath: "prompts/nested/user.tmpl ",
wantDisplay: "nested/user.tmpl ",
},
{ {
name: "escape rejected", name: "escape rejected",
root: "prompts", root: "prompts",
@@ -218,26 +237,6 @@ func TestResolveFSPath(t *testing.T) {
} }
} }
func TestStemStripsYAMLExtensions(t *testing.T) {
tests := []struct {
name string
in string
want string
}{
{name: "yaml", in: "prompt.yaml", want: "prompt"},
{name: "yml", in: "profile.yml", want: "profile"},
{name: "other", in: "file.txt", want: "file.txt"},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
if got := Stem(tc.in); got != tc.want {
t.Fatalf("expected %q, got %q", tc.want, got)
}
})
}
}
func TestIsYAMLFile(t *testing.T) { func TestIsYAMLFile(t *testing.T) {
tests := []struct { tests := []struct {
name string name string

View File

@@ -1,5 +1,5 @@
// Package jsonvalue validates and defensively copies JSON-compatible value // Package jsonvalue validates and defensively copies bounded JSON-compatible
// trees used by public configuration and request boundaries. // value trees used by configuration, request, and prepared-state boundaries.
package jsonvalue package jsonvalue
import ( import (
@@ -8,23 +8,39 @@ import (
"math" "math"
"reflect" "reflect"
"sort" "sort"
"strconv"
) )
const maxSafeJSONInteger = 1<<53 - 1 const (
maxContainerDepth = 100
maxProducedNodes = 100_000
)
type visit struct { type visit struct {
typ reflect.Type typ reflect.Type
ptr uintptr ptr uintptr
} }
type traversalState struct {
active map[visit]struct{}
producedNodes int
}
// Copy validates and deeply copies a JSON-compatible value while preserving
// compatible concrete map, slice, array, scalar, and number types. It rejects
// cycles and values that exceed the package's traversal limits.
func Copy(src any) (any, error) {
return copyValue(reflect.ValueOf(src), "value", newTraversalState(), true, 0)
}
// CopyMap validates and deeply copies an extra-parameter map while preserving // CopyMap validates and deeply copies an extra-parameter map while preserving
// compatible concrete map, slice, array, scalar, and number types. // compatible concrete map, slice, array, scalar, and number types. It rejects
// empty object keys, cycles, and values that exceed the package's traversal
// limits.
func CopyMap(src map[string]any) (map[string]any, error) { func CopyMap(src map[string]any) (map[string]any, error) {
if src == nil { if src == nil {
return nil, nil return nil, nil
} }
copied, err := copyValue(reflect.ValueOf(src), "extra_params", make(map[visit]struct{})) copied, err := copyValue(reflect.ValueOf(src), "extra_params", newTraversalState(), false, 0)
if err != nil { if err != nil {
return nil, err return nil, err
} }
@@ -35,88 +51,121 @@ func CopyMap(src map[string]any) (map[string]any, error) {
return out, nil return out, nil
} }
func copyValue(value reflect.Value, path string, seen map[visit]struct{}) (any, error) { func copyValue(
if !value.IsValid() { value reflect.Value,
path string,
state *traversalState,
allowEmptyMapKeys bool,
containerDepth int,
) (any, error) {
resolved, cleanup, isNull, err := state.resolveIndirection(value, path)
if err != nil {
return nil, err
}
defer cleanup()
if isNull {
if err := state.produceNode(path); err != nil {
return nil, err
}
return nil, nil return nil, nil
} }
if value.Kind() == reflect.Interface { value = resolved
if value.IsNil() {
return nil, nil
}
return copyValue(value.Elem(), path, seen)
}
if !value.CanInterface() { if !value.CanInterface() {
return nil, fmt.Errorf("%s: value cannot be copied", path) return nil, fmt.Errorf("%s: value cannot be copied", path)
} }
if number, ok := value.Interface().(json.Number); ok { if number, ok := value.Interface().(json.Number); ok {
if _, err := json.Marshal(number); err != nil { if !validJSONNumber(number) {
return nil, fmt.Errorf("%s: invalid JSON number", path) return nil, fmt.Errorf("%s: invalid JSON number", path)
} }
f, err := strconv.ParseFloat(number.String(), 64) if err := state.produceNode(path); err != nil {
if err != nil || math.IsNaN(f) || math.IsInf(f, 0) { return nil, err
return nil, fmt.Errorf("%s: invalid JSON number", path)
} }
return number, nil return number, nil
} }
switch value.Kind() { switch value.Kind() {
case reflect.Bool, reflect.String: case reflect.Bool, reflect.String:
if err := state.produceNode(path); err != nil {
return nil, err
}
return value.Interface(), nil return value.Interface(), nil
case reflect.Int, reflect.Int8, reflect.Int16, reflect.Int32, reflect.Int64: case reflect.Int, reflect.Int8, reflect.Int16, reflect.Int32, reflect.Int64:
if value.Int() < -maxSafeJSONInteger || value.Int() > maxSafeJSONInteger { if err := state.produceNode(path); err != nil {
return nil, fmt.Errorf("%s: integer is outside the JSON-safe range", path) return nil, err
} }
return value.Interface(), nil return value.Interface(), nil
case reflect.Uint, reflect.Uint8, reflect.Uint16, reflect.Uint32, reflect.Uint64, reflect.Uintptr: case reflect.Uint, reflect.Uint8, reflect.Uint16, reflect.Uint32, reflect.Uint64, reflect.Uintptr:
if value.Uint() > maxSafeJSONInteger { if err := state.produceNode(path); err != nil {
return nil, fmt.Errorf("%s: integer is outside the JSON-safe range", path) return nil, err
} }
return value.Interface(), nil return value.Interface(), nil
case reflect.Float32, reflect.Float64: case reflect.Float32, reflect.Float64:
number := value.Convert(reflect.TypeOf(float64(0))).Float() number := value.Float()
if math.IsNaN(number) || math.IsInf(number, 0) { if math.IsNaN(number) || math.IsInf(number, 0) {
return nil, fmt.Errorf("%s: floating-point value must be finite", path) return nil, fmt.Errorf("%s: floating-point value must be finite", path)
} }
if err := state.produceNode(path); err != nil {
return nil, err
}
return value.Interface(), nil return value.Interface(), nil
case reflect.Pointer: case reflect.Map:
if value.IsNil() { if value.IsNil() {
if err := state.produceNode(path); err != nil {
return nil, err
}
return nil, nil return nil, nil
} }
current := visit{typ: value.Type(), ptr: value.Pointer()} nextDepth, err := state.enterContainer(path, containerDepth)
if _, ok := seen[current]; ok { if err != nil {
return nil, fmt.Errorf("%s: cyclic value is not supported", path) return nil, err
} }
seen[current] = struct{}{} return copyMapValue(value, path, state, allowEmptyMapKeys, nextDepth)
defer delete(seen, current)
return copyValue(value.Elem(), path, seen)
case reflect.Map:
return copyMapValue(value, path, seen)
case reflect.Slice: case reflect.Slice:
if value.IsNil() { if value.IsNil() {
if err := state.produceNode(path); err != nil {
return nil, err
}
return nil, nil return nil, nil
} }
return copySequenceValue(value, path, seen) nextDepth, err := state.enterContainer(path, containerDepth)
if err != nil {
return nil, err
}
return copySequenceValue(value, path, state, allowEmptyMapKeys, nextDepth)
case reflect.Array: case reflect.Array:
return copySequenceValue(value, path, seen) nextDepth, err := state.enterContainer(path, containerDepth)
if err != nil {
return nil, err
}
return copySequenceValue(value, path, state, allowEmptyMapKeys, nextDepth)
default: default:
return nil, fmt.Errorf("%s: unsupported JSON value type %s", path, value.Type()) return nil, fmt.Errorf("%s: unsupported JSON value type %s", path, value.Type())
} }
} }
func copyMapValue(value reflect.Value, path string, seen map[visit]struct{}) (any, error) { func copyMapValue(
if value.IsNil() { value reflect.Value,
return nil, nil path string,
} state *traversalState,
allowEmptyMapKeys bool,
containerDepth int,
) (any, error) {
if value.Type().Key().Kind() != reflect.String { if value.Type().Key().Kind() != reflect.String {
return nil, fmt.Errorf("%s: map key type %s is not supported", path, value.Type().Key()) return nil, fmt.Errorf("%s: map key type %s is not supported", path, value.Type().Key())
} }
if err := state.produceNode(path); err != nil {
return nil, err
}
if err := state.ensureChildCapacity(path, value.Len()); err != nil {
return nil, err
}
current := visit{typ: value.Type(), ptr: value.Pointer()} current := visit{typ: value.Type(), ptr: value.Pointer()}
if _, ok := seen[current]; ok { if _, ok := state.active[current]; ok {
return nil, fmt.Errorf("%s: cyclic value is not supported", path) return nil, fmt.Errorf("%s: cyclic value is not supported", path)
} }
seen[current] = struct{}{} state.active[current] = struct{}{}
defer delete(seen, current) defer delete(state.active, current)
keys := value.MapKeys() keys := value.MapKeys()
sort.Slice(keys, func(i, j int) bool { sort.Slice(keys, func(i, j int) bool {
@@ -133,10 +182,16 @@ func copyMapValue(value reflect.Value, path string, seen map[visit]struct{}) (an
elementType := value.Type().Elem() elementType := value.Type().Elem()
for _, key := range keys { for _, key := range keys {
name := key.String() name := key.String()
if name == "" { if name == "" && !allowEmptyMapKeys {
return nil, fmt.Errorf("%s: map key must not be empty", path) return nil, fmt.Errorf("%s: map key must not be empty", path)
} }
copied, err := copyValue(value.MapIndex(key), path+"."+name, seen) copied, err := copyValue(
value.MapIndex(key),
path+"."+name,
state,
allowEmptyMapKeys,
containerDepth,
)
if err != nil { if err != nil {
return nil, err return nil, err
} }
@@ -171,22 +226,41 @@ func copyMapValue(value reflect.Value, path string, seen map[visit]struct{}) (an
return out, nil return out, nil
} }
func copySequenceValue(value reflect.Value, path string, seen map[visit]struct{}) (any, error) { func copySequenceValue(
value reflect.Value,
path string,
state *traversalState,
allowEmptyMapKeys bool,
containerDepth int,
) (any, error) {
if err := state.produceNode(path); err != nil {
return nil, err
}
if err := state.ensureChildCapacity(path, value.Len()); err != nil {
return nil, err
}
var current visit var current visit
if value.Kind() == reflect.Slice { if value.Kind() == reflect.Slice {
current = visit{typ: value.Type(), ptr: value.Pointer()} current = visit{typ: value.Type(), ptr: value.Pointer()}
if _, ok := seen[current]; ok { if _, ok := state.active[current]; ok {
return nil, fmt.Errorf("%s: cyclic value is not supported", path) return nil, fmt.Errorf("%s: cyclic value is not supported", path)
} }
seen[current] = struct{}{} state.active[current] = struct{}{}
defer delete(seen, current) defer delete(state.active, current)
} }
values := make([]any, value.Len()) values := make([]any, value.Len())
preserveType := true preserveType := true
elementType := value.Type().Elem() elementType := value.Type().Elem()
for i := 0; i < value.Len(); i++ { for i := 0; i < value.Len(); i++ {
copied, err := copyValue(value.Index(i), fmt.Sprintf("%s[%d]", path, i), seen) copied, err := copyValue(
value.Index(i),
fmt.Sprintf("%s[%d]", path, i),
state,
allowEmptyMapKeys,
containerDepth,
)
if err != nil { if err != nil {
return nil, err return nil, err
} }
@@ -222,6 +296,73 @@ func copySequenceValue(value reflect.Value, path string, seen map[visit]struct{}
return out, nil return out, nil
} }
func validJSONNumber(number json.Number) bool {
var parsed json.Number
if err := json.Unmarshal([]byte(number.String()), &parsed); err != nil {
return false
}
return parsed.String() == number.String()
}
func newTraversalState() *traversalState {
return &traversalState{active: make(map[visit]struct{})}
}
func (state *traversalState) produceNode(path string) error {
if state.producedNodes >= maxProducedNodes {
return fmt.Errorf("%s: JSON value work limit exceeded", path)
}
state.producedNodes++
return nil
}
func (state *traversalState) enterContainer(path string, depth int) (int, error) {
depth++
if depth > maxContainerDepth {
return 0, fmt.Errorf("%s: JSON container depth limit exceeded", path)
}
return depth, nil
}
func (state *traversalState) ensureChildCapacity(path string, count int) error {
if count > maxProducedNodes-state.producedNodes {
return fmt.Errorf("%s: JSON value work limit exceeded", path)
}
return nil
}
func (state *traversalState) resolveIndirection(
value reflect.Value,
path string,
) (reflect.Value, func(), bool, error) {
var visits []visit
cleanup := func() {
for _, current := range visits {
delete(state.active, current)
}
}
for value.IsValid() && (value.Kind() == reflect.Interface || value.Kind() == reflect.Pointer) {
if value.IsNil() {
return reflect.Value{}, cleanup, true, nil
}
if value.Kind() == reflect.Pointer {
current := visit{typ: value.Type(), ptr: value.Pointer()}
if _, ok := state.active[current]; ok {
cleanup()
return reflect.Value{}, nil, false, fmt.Errorf("%s: cyclic value is not supported", path)
}
state.active[current] = struct{}{}
visits = append(visits, current)
}
value = value.Elem()
}
if !value.IsValid() {
return reflect.Value{}, cleanup, true, nil
}
return value, cleanup, false, nil
}
func canAssignNil(typ reflect.Type) bool { func canAssignNil(typ reflect.Type) bool {
switch typ.Kind() { switch typ.Kind() {
case reflect.Chan, reflect.Func, reflect.Interface, reflect.Map, reflect.Pointer, reflect.Slice: case reflect.Chan, reflect.Func, reflect.Interface, reflect.Map, reflect.Pointer, reflect.Slice:

View File

@@ -1,97 +1,332 @@
package jsonvalue_test package jsonvalue
import ( import (
"encoding/json" "encoding/json"
"math" "math"
"reflect" "reflect"
"strings"
"testing" "testing"
"gitea.maximumdirect.net/eric/promptkit/internal/jsonvalue"
) )
func TestCopyMapPreservesTypesAndIsolatesMutations(t *testing.T) { type (
nested := map[string]int{"limit": 2} namedBool bool
sequence := []string{"one", "two"} namedString string
input := map[string]any{ namedInt64 int64
"count": int64(7), namedUint64 uint64
"number": json.Number("-1.25e+2"), namedFloat32 float32
"nested": nested, namedFloat64 float64
"sequence": sequence, namedKey string
} namedMap map[namedKey]namedInt64
namedSlice []namedString
copied, err := jsonvalue.CopyMap(input) namedArray [1]map[string]int
if err != nil { )
t.Fatalf("copy map: %v", err)
}
nested["limit"] = 99
sequence[0] = "changed"
input["added"] = true
if got, ok := copied["count"].(int64); !ok || got != 7 {
t.Fatalf("integer type or value changed: %#v", copied["count"])
}
if got, ok := copied["number"].(json.Number); !ok || got != "-1.25e+2" {
t.Fatalf("JSON number type or value changed: %#v", copied["number"])
}
if got := copied["nested"].(map[string]int)["limit"]; got != 2 {
t.Fatalf("nested map was not isolated: %d", got)
}
if got := copied["sequence"].([]string)[0]; got != "one" {
t.Fatalf("sequence was not isolated: %q", got)
}
if _, ok := copied["added"]; ok {
t.Fatalf("top-level map was not isolated: %#v", copied)
}
}
func TestCopyMapRejectsInvalidValues(t *testing.T) {
cyclicMap := map[string]any{}
cyclicMap["self"] = cyclicMap
cyclicSlice := []any{nil}
cyclicSlice[0] = cyclicSlice
func TestCopyPreservesSupportedScalarAndNumberTypes(t *testing.T) {
maxInt := int(^uint(0) >> 1)
minInt := -maxInt - 1
tests := []struct { tests := []struct {
name string name string
value any value any
}{ }{
{name: "empty nested key", value: map[string]int{"": 1}}, {name: "bool", value: true},
{name: "non-string map key", value: map[int]string{1: "one"}}, {name: "named bool", value: namedBool(true)},
{name: "unsupported value", value: make(chan int)}, {name: "string", value: "value"},
{name: "cyclic map", value: cyclicMap}, {name: "named string", value: namedString("value")},
{name: "cyclic slice", value: cyclicSlice}, {name: "int", value: minInt},
{name: "NaN", value: math.NaN()}, {name: "int8", value: int8(-1 << 7)},
{name: "positive infinity", value: math.Inf(1)}, {name: "int16", value: int16(-1 << 15)},
{name: "unsafe signed integer", value: int64(1 << 53)}, {name: "int32", value: int32(-1 << 31)},
{name: "unsafe unsigned integer", value: uint64(1 << 53)}, {name: "int64", value: int64(-1 << 63)},
{name: "named int64", value: namedInt64(1<<63 - 1)},
{name: "uint", value: ^uint(0)},
{name: "uint8", value: ^uint8(0)},
{name: "uint16", value: ^uint16(0)},
{name: "uint32", value: ^uint32(0)},
{name: "uint64", value: ^uint64(0)},
{name: "uintptr", value: ^uintptr(0)},
{name: "named uint64", value: namedUint64(^uint64(0))},
{name: "float32", value: float32(1.25)},
{name: "float64", value: float64(-2.5e100)},
{name: "named float32", value: namedFloat32(3.5)},
{name: "named float64", value: namedFloat64(-4.5e200)},
{name: "JSON number integer", value: json.Number("18446744073709551615")},
{name: "JSON number fraction", value: json.Number("-1.25e+2")},
{name: "JSON number beyond float64", value: json.Number("1e9999")},
} }
for _, tc := range tests { for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) { t.Run(tc.name, func(t *testing.T) {
if _, err := jsonvalue.CopyMap(map[string]any{"value": tc.value}); err == nil { got, err := Copy(tc.value)
if err != nil {
t.Fatalf("copy value: %v", err)
}
if !reflect.DeepEqual(got, tc.value) {
t.Fatalf("value or concrete type changed: got %#v (%T), want %#v (%T)", got, got, tc.value, tc.value)
}
})
}
}
func TestCopyRejectsInvalidNumbers(t *testing.T) {
tests := []struct {
name string
value any
}{
{name: "float32 NaN", value: float32(math.NaN())},
{name: "float64 NaN", value: math.NaN()},
{name: "named float NaN", value: namedFloat64(math.NaN())},
{name: "positive infinity", value: math.Inf(1)},
{name: "negative infinity", value: math.Inf(-1)},
{name: "empty JSON number", value: json.Number("")},
{name: "leading zero JSON number", value: json.Number("01")},
{name: "leading plus JSON number", value: json.Number("+1")},
{name: "trailing decimal JSON number", value: json.Number("1.")},
{name: "leading decimal JSON number", value: json.Number(".1")},
{name: "non-number JSON number", value: json.Number("NaN")},
{name: "spaced JSON number", value: json.Number(" 1")},
{name: "quoted JSON number", value: json.Number(`"1"`)},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
if _, err := Copy(tc.value); err == nil {
t.Fatal("expected validation error") t.Fatal("expected validation error")
} }
}) })
} }
} }
func TestCopyMapValidatesJSONNumberSyntaxAndRange(t *testing.T) { func TestCopyPreservesCompatibleCollectionsAndNilEmptyDistinctions(t *testing.T) {
for _, number := range []json.Number{"0", "-1", "1.25", "-1.25e+2"} { collections := []struct {
t.Run("valid "+number.String(), func(t *testing.T) { name string
got, err := jsonvalue.CopyMap(map[string]any{"value": number}) value any
}{
{name: "unnamed map", value: map[string]int{"limit": 2}},
{name: "named map", value: namedMap{"limit": 2}},
{name: "unnamed slice", value: []string{"one", "two"}},
{name: "named slice", value: namedSlice{"one", "two"}},
{name: "unnamed array", value: [2]int{1, 2}},
{name: "named array", value: namedArray{{"limit": 2}}},
{name: "empty map", value: map[string]int{}},
{name: "empty named map", value: namedMap{}},
{name: "empty slice", value: []string{}},
{name: "empty named slice", value: namedSlice{}},
{name: "empty array", value: [0]string{}},
}
for _, tc := range collections {
t.Run(tc.name, func(t *testing.T) {
got, err := Copy(tc.value)
if err != nil { if err != nil {
t.Fatalf("copy valid JSON number: %v", err) t.Fatalf("copy collection: %v", err)
} }
if !reflect.DeepEqual(got["value"], number) { if !reflect.DeepEqual(got, tc.value) || reflect.TypeOf(got) != reflect.TypeOf(tc.value) {
t.Fatalf("JSON number changed: got %#v want %#v", got["value"], number) t.Fatalf("collection changed: got %#v (%T), want %#v (%T)", got, got, tc.value, tc.value)
}
kind := reflect.ValueOf(got).Kind()
if (kind == reflect.Map || kind == reflect.Slice) && reflect.ValueOf(got).IsNil() {
t.Fatal("non-nil collection became nil")
} }
}) })
} }
for _, number := range []json.Number{"", "01", "+1", "1.", ".1", "1e9999", "not-a-number"} { var nilMap map[string]int
t.Run("invalid "+number.String(), func(t *testing.T) { var nilSlice []string
if _, err := jsonvalue.CopyMap(map[string]any{"value": number}); err == nil { var nilPointer *namedInt64
t.Fatal("expected invalid JSON number error") for _, value := range []any{nil, nilMap, nilSlice, nilPointer} {
got, err := Copy(value)
if err != nil {
t.Fatalf("copy null value: %v", err)
}
if got != nil {
t.Fatalf("null value became %#v (%T)", got, got)
}
}
gotNil, err := CopyMap(nil)
if err != nil || gotNil != nil {
t.Fatalf("nil CopyMap result = %#v, %v", gotNil, err)
}
gotEmpty, err := CopyMap(map[string]any{})
if err != nil || gotEmpty == nil || len(gotEmpty) != 0 {
t.Fatalf("empty CopyMap result = %#v, %v", gotEmpty, err)
}
}
func TestCopyHandlesIndirectionAndIsolatesNestedMutations(t *testing.T) {
integer := namedInt64(7)
nestedMap := namedMap{"limit": 2}
nestedSlice := namedSlice{"original"}
nestedArray := namedArray{{"limit": 3}}
shared := []any{map[string]int{"value": 4}}
input := map[string]any{
"integer": &integer,
"map": nestedMap,
"slice": nestedSlice,
"array": nestedArray,
"first": shared,
"second": shared,
}
copiedValue, err := Copy(input)
if err != nil {
t.Fatalf("copy mixed tree: %v", err)
}
copied := copiedValue.(map[string]any)
nestedMap["limit"] = 20
nestedSlice[0] = "changed"
nestedArray[0]["limit"] = 30
shared[0].(map[string]int)["value"] = 40
if got, ok := copied["integer"].(namedInt64); !ok || got != 7 {
t.Fatalf("pointer target changed: %#v", copied["integer"])
}
if got := copied["map"].(namedMap)["limit"]; got != 2 {
t.Fatalf("nested map aliased input: %d", got)
}
if got := copied["slice"].(namedSlice)[0]; got != "original" {
t.Fatalf("nested slice aliased input: %q", got)
}
if got := copied["array"].(namedArray)[0]["limit"]; got != 3 {
t.Fatalf("nested array aliased input: %d", got)
}
first := copied["first"].([]any)
second := copied["second"].([]any)
if got := first[0].(map[string]int)["value"]; got != 4 {
t.Fatalf("shared child aliased input: %d", got)
}
first[0].(map[string]int)["value"] = 99
if got := second[0].(map[string]int)["value"]; got != 4 {
t.Fatalf("repeated acyclic value shared copied output: %d", got)
}
}
func TestCopyAndCopyMapApplyDistinctEmptyKeyRules(t *testing.T) {
nested := map[string]any{"": []any{"original"}}
copiedValue, err := Copy(nested)
if err != nil {
t.Fatalf("Copy rejected empty schema key: %v", err)
}
nested[""].([]any)[0] = "changed"
if got := copiedValue.(map[string]any)[""].([]any)[0]; got != "original" {
t.Fatalf("copied schema value was not isolated: %v", got)
}
_, err = CopyMap(map[string]any{"nested": map[string]any{"": true}})
if err == nil || !strings.Contains(err.Error(), "extra_params.nested") {
t.Fatalf("CopyMap empty-key error = %v", err)
}
}
func TestCopyRejectsUnsupportedValuesAndActiveCycles(t *testing.T) {
cyclicMap := map[string]any{}
cyclicMap["self"] = cyclicMap
cyclicSlice := []any{nil}
cyclicSlice[0] = cyclicSlice
var cyclicPointer any
cyclicPointer = &cyclicPointer
tests := []struct {
name string
value any
wantPath string
}{
{name: "non-string map key", value: map[int]string{1: "one"}, wantPath: "value"},
{name: "unsupported channel", value: make(chan int), wantPath: "value"},
{name: "deterministic map path", value: map[string]any{"z": make(chan int), "a": make(chan int)}, wantPath: "value.a"},
{name: "cyclic map", value: cyclicMap, wantPath: "value.self"},
{name: "cyclic slice", value: cyclicSlice, wantPath: "value[0]"},
{name: "cyclic pointer", value: cyclicPointer, wantPath: "value"},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
_, err := Copy(tc.value)
if err == nil || !strings.Contains(err.Error(), tc.wantPath) {
t.Fatalf("error = %v, want structural path %q", err, tc.wantPath)
} }
}) })
} }
} }
func TestCopyEnforcesContainerDepth(t *testing.T) {
tests := []struct {
name string
depth int
wantErr bool
}{
{name: "just below", depth: maxContainerDepth - 1},
{name: "at limit", depth: maxContainerDepth},
{name: "over limit", depth: maxContainerDepth + 1, wantErr: true},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
_, err := Copy(alternatingContainers(tc.depth))
if tc.wantErr {
if err == nil || !strings.HasPrefix(err.Error(), "value") || !strings.Contains(err.Error(), "container depth limit") {
t.Fatalf("depth error = %v", err)
}
return
}
if err != nil {
t.Fatalf("copy depth %d: %v", tc.depth, err)
}
})
}
}
func TestCopyEnforcesProducedNodeBudgetForRepeatedAcyclicValues(t *testing.T) {
shared := []any{true}
sharedOccurrences := (maxProducedNodes - 2) / 2
justBelow := repeatedValues(shared, sharedOccurrences, 0)
atLimit := repeatedValues(shared, sharedOccurrences, 1)
overLimit := repeatedValues(shared, sharedOccurrences, 2)
for name, value := range map[string]any{
"just below": justBelow,
"at limit": atLimit,
} {
t.Run(name, func(t *testing.T) {
if _, err := Copy(value); err != nil {
t.Fatalf("copy value within work budget: %v", err)
}
})
}
_, err := Copy(overLimit)
if err == nil || !strings.HasPrefix(err.Error(), "value[") || !strings.Contains(err.Error(), "value work limit") {
t.Fatalf("work-budget error = %v", err)
}
_, err = Copy(make([]any, maxProducedNodes))
if err == nil || !strings.HasPrefix(err.Error(), "value:") || !strings.Contains(err.Error(), "value work limit") {
t.Fatalf("flat work-budget error = %v", err)
}
}
func alternatingContainers(depth int) any {
var value any = true
for level := 0; level < depth; level++ {
switch level % 3 {
case 0:
value = map[string]any{"child": value}
case 1:
value = []any{value}
default:
value = [1]any{value}
}
}
return value
}
func repeatedValues(shared []any, occurrences, leadingScalars int) []any {
values := make([]any, 0, leadingScalars+occurrences)
for i := 0; i < leadingScalars; i++ {
values = append(values, false)
}
for i := 0; i < occurrences; i++ {
values = append(values, shared)
}
return values
}

View File

@@ -25,6 +25,20 @@ var (
ErrMalformedResponse = errors.New("malformed llm response") ErrMalformedResponse = errors.New("malformed llm response")
) )
const maxOpenAIChatResponseBytes int64 = 16 << 20
type requestFailedError struct {
cause error
}
func (e *requestFailedError) Error() string {
return ErrRequestFailed.Error()
}
func (e *requestFailedError) Unwrap() []error {
return []error{ErrRequestFailed, e.cause}
}
type OpenAICompatibleConfig struct { type OpenAICompatibleConfig struct {
BaseURL string BaseURL string
Model string Model string
@@ -39,9 +53,11 @@ type OpenAICompatibleClient struct {
} }
func NewOpenAICompatibleClient(cfg OpenAICompatibleConfig) (*OpenAICompatibleClient, error) { func NewOpenAICompatibleClient(cfg OpenAICompatibleConfig) (*OpenAICompatibleClient, error) {
baseURL := strings.TrimSpace(cfg.BaseURL) baseURL := ""
if baseURL != "" { if strings.TrimSpace(cfg.BaseURL) != "" {
if _, err := url.ParseRequestURI(baseURL); err != nil { var err error
baseURL, err = domain.NormalizeOpenAICompatibleBaseEndpoint(cfg.BaseURL)
if err != nil {
return nil, fmt.Errorf("%w: invalid base URL: %v", ErrInvalidConfig, err) return nil, fmt.Errorf("%w: invalid base URL: %v", ErrInvalidConfig, err)
} }
} }
@@ -63,25 +79,29 @@ func NewOpenAICompatibleClient(cfg OpenAICompatibleConfig) (*OpenAICompatibleCli
} }
return &OpenAICompatibleClient{ return &OpenAICompatibleClient{
baseURL: strings.TrimRight(baseURL, "/"), baseURL: baseURL,
defaultModel: cfg.Model, defaultModel: cfg.Model,
httpClient: client, httpClient: client,
}, nil }, nil
} }
func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.GenerateRequest) (*domain.GenerateResponse, error) { func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.GenerateRequest) (*domain.GenerateResponse, error) {
if req.Target.TimeoutSeconds < 0 { if err := domain.ValidateExecutionTargetSettings(req.Target); err != nil {
return nil, fmt.Errorf("%w: timeout_seconds must be greater than or equal to 0", ErrInvalidRequest) return nil, fmt.Errorf("%w: %v", ErrInvalidRequest, err)
} }
endpoint := strings.TrimSpace(req.Target.Endpoint) selectedEndpoint := req.Target.Endpoint
if endpoint == "" { if strings.TrimSpace(selectedEndpoint) == "" {
endpoint = c.baseURL selectedEndpoint = c.baseURL
} }
if endpoint == "" { endpoint, err := domain.NormalizeOpenAICompatibleBaseEndpoint(selectedEndpoint)
return nil, fmt.Errorf("%w: endpoint is required", ErrInvalidRequest) if err != nil {
return nil, fmt.Errorf("%w: invalid endpoint: %v", ErrInvalidRequest, err)
}
endpoint, err = url.JoinPath(endpoint, defaults.OpenAIChatCompletionsPath)
if err != nil {
return nil, fmt.Errorf("%w: invalid endpoint path: %v", ErrInvalidRequest, err)
} }
endpoint = strings.TrimRight(endpoint, "/") + defaults.OpenAIChatCompletionsPath
wireReq, err := openAIChatRequestFromGenerateRequest(req, c.defaultModel) wireReq, err := openAIChatRequestFromGenerateRequest(req, c.defaultModel)
if err != nil { if err != nil {
@@ -113,13 +133,18 @@ func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.Genera
return nil, fmt.Errorf("%w: failed to create request: %v", ErrRequestFailed, err) return nil, fmt.Errorf("%w: failed to create request: %v", ErrRequestFailed, err)
} }
httpReq.Header.Set("Content-Type", "application/json") httpReq.Header.Set("Content-Type", "application/json")
if apiKey := strings.TrimSpace(req.Target.APIKey); apiKey != "" { apiKey := strings.TrimSpace(req.Target.APIKey)
httpReq.Header.Set("Authorization", "Bearer "+apiKey) envName := strings.TrimSpace(req.Target.APIKeyEnv)
} else if envName := strings.TrimSpace(req.Target.APIKeyEnv); envName != "" { if apiKey == "" && envName != "" {
apiKey := strings.TrimSpace(os.Getenv(envName)) apiKey = strings.TrimSpace(os.Getenv(envName))
if apiKey == "" { }
if apiKey == "" && req.Target.APIKeyRequired {
if envName != "" {
return nil, fmt.Errorf("%w: api key environment variable %q is not set", ErrInvalidRequest, envName) return nil, fmt.Errorf("%w: api key environment variable %q is not set", ErrInvalidRequest, envName)
} }
return nil, fmt.Errorf("%w: api key is required", ErrInvalidRequest)
}
if apiKey != "" {
httpReq.Header.Set("Authorization", "Bearer "+apiKey) httpReq.Header.Set("Authorization", "Bearer "+apiKey)
} }
@@ -130,30 +155,36 @@ func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.Genera
httpResp, err := httpClient.Do(httpReq) httpResp, err := httpClient.Do(httpReq)
if err != nil { if err != nil {
return nil, fmt.Errorf("%w: %v", ErrRequestFailed, err) return nil, &requestFailedError{cause: err}
} }
defer httpResp.Body.Close() defer httpResp.Body.Close()
if httpResp.StatusCode < 200 || httpResp.StatusCode >= 300 { if httpResp.StatusCode < 200 || httpResp.StatusCode >= 300 {
_, _ = io.Copy(io.Discard, io.LimitReader(httpResp.Body, 4096)) return nil, providerHTTPErrorFromBody(
return nil, fmt.Errorf("%w: status=%d", ErrUnexpectedStatus, httpResp.StatusCode) httpResp.StatusCode,
httpResp.ContentLength,
httpResp.Body,
)
}
if httpResp.ContentLength > maxOpenAIChatResponseBytes {
return nil, openAIChatResponseTooLargeError()
} }
var wireResp openAIChatResponse wireResp, err := decodeOpenAIChatResponse(httpResp.Body)
if err := json.NewDecoder(httpResp.Body).Decode(&wireResp); err != nil { if err != nil {
return nil, fmt.Errorf("%w: failed to decode response: %v", ErrMalformedResponse, err) return nil, err
} }
if len(wireResp.Choices) == 0 { if len(wireResp.Choices) == 0 {
return nil, fmt.Errorf("%w: no choices returned", ErrMalformedResponse) return nil, fmt.Errorf("%w: no choices returned", ErrMalformedResponse)
} }
content := wireResp.Choices[0].Message.Content content := wireResp.Choices[0].Message.Content
if content == "" { if content == nil {
return nil, fmt.Errorf("%w: first choice has empty message content", ErrMalformedResponse) return nil, fmt.Errorf("%w: first choice has missing message content", ErrMalformedResponse)
} }
return &domain.GenerateResponse{ return &domain.GenerateResponse{
Content: content, Content: *content,
Usage: domain.TokenUsage{ Usage: domain.TokenUsage{
PromptTokens: wireResp.Usage.PromptTokens, PromptTokens: wireResp.Usage.PromptTokens,
CompletionTokens: wireResp.Usage.CompletionTokens, CompletionTokens: wireResp.Usage.CompletionTokens,
@@ -164,6 +195,46 @@ func (c *OpenAICompatibleClient) Generate(ctx context.Context, req domain.Genera
}, nil }, nil
} }
func decodeOpenAIChatResponse(body io.Reader) (openAIChatResponse, error) {
limited := &io.LimitedReader{
R: body,
N: maxOpenAIChatResponseBytes + 1,
}
decoder := json.NewDecoder(limited)
var response openAIChatResponse
if err := decoder.Decode(&response); err != nil {
if limited.N == 0 {
return openAIChatResponse{}, openAIChatResponseTooLargeError()
}
return openAIChatResponse{}, fmt.Errorf("%w: failed to decode response", ErrMalformedResponse)
}
if limited.N == 0 {
return openAIChatResponse{}, openAIChatResponseTooLargeError()
}
var trailing any
if err := decoder.Decode(&trailing); !errors.Is(err, io.EOF) {
if limited.N == 0 {
return openAIChatResponse{}, openAIChatResponseTooLargeError()
}
return openAIChatResponse{}, fmt.Errorf("%w: response contains trailing data", ErrMalformedResponse)
}
if limited.N == 0 {
return openAIChatResponse{}, openAIChatResponseTooLargeError()
}
return response, nil
}
func openAIChatResponseTooLargeError() error {
return fmt.Errorf(
"%w: response exceeds %d-byte limit",
ErrMalformedResponse,
maxOpenAIChatResponseBytes,
)
}
func openAIChatRequestFromGenerateRequest(req domain.GenerateRequest, defaultModel string) (openAIChatRequest, error) { func openAIChatRequestFromGenerateRequest(req domain.GenerateRequest, defaultModel string) (openAIChatRequest, error) {
model := strings.TrimSpace(req.Target.Model) model := strings.TrimSpace(req.Target.Model)
if model == "" { if model == "" {
@@ -308,8 +379,8 @@ type openAICacheControl struct {
} }
type openAIChatResponseMessage struct { type openAIChatResponseMessage struct {
Role string `json:"role"` Role string `json:"role"`
Content string `json:"content"` Content *string `json:"content"`
} }
type openAIChatResponse struct { type openAIChatResponse struct {

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,189 @@
package llm
import (
"encoding/json"
"errors"
"fmt"
"io"
"strings"
"unicode"
)
const (
maxProviderErrorResponseBytes int64 = 64 << 10
maxProviderErrorIdentifierRunes = 256
maxProviderErrorMessageRunes = 4096
)
// ProviderHTTPError describes a non-success response from an LLM provider.
type ProviderHTTPError struct {
statusCode int
providerCode string
providerType string
providerMessage string
}
func (e *ProviderHTTPError) StatusCode() int {
if e == nil {
return 0
}
return e.statusCode
}
func (e *ProviderHTTPError) ProviderCode() string {
if e == nil {
return ""
}
return e.providerCode
}
func (e *ProviderHTTPError) ProviderType() string {
if e == nil {
return ""
}
return e.providerType
}
func (e *ProviderHTTPError) ProviderMessage() string {
if e == nil {
return ""
}
return e.providerMessage
}
func (e *ProviderHTTPError) Error() string {
if e == nil || e.statusCode == 0 {
return ErrUnexpectedStatus.Error()
}
return fmt.Sprintf("%s: status=%d", ErrUnexpectedStatus, e.statusCode)
}
func (e *ProviderHTTPError) GoString() string {
return e.Error()
}
func (e *ProviderHTTPError) Unwrap() error {
return ErrUnexpectedStatus
}
type providerErrorDetails struct {
providerCode string
providerType string
providerMessage string
}
func newProviderHTTPError(statusCode int, details providerErrorDetails) *ProviderHTTPError {
return &ProviderHTTPError{
statusCode: statusCode,
providerCode: details.providerCode,
providerType: details.providerType,
providerMessage: details.providerMessage,
}
}
func providerHTTPErrorFromBody(statusCode int, contentLength int64, body io.Reader) *ProviderHTTPError {
if contentLength > maxProviderErrorResponseBytes {
return newProviderHTTPError(statusCode, providerErrorDetails{})
}
limited := &io.LimitedReader{
R: body,
N: maxProviderErrorResponseBytes + 1,
}
contents, err := io.ReadAll(limited)
if err != nil || limited.N == 0 {
return newProviderHTTPError(statusCode, providerErrorDetails{})
}
return newProviderHTTPError(statusCode, parseProviderErrorEnvelope(contents))
}
func parseProviderErrorEnvelope(body []byte) providerErrorDetails {
decoder := json.NewDecoder(strings.NewReader(string(body)))
decoder.UseNumber()
var envelope map[string]json.RawMessage
if err := decoder.Decode(&envelope); err != nil {
return providerErrorDetails{}
}
var trailing any
if err := decoder.Decode(&trailing); !errors.Is(err, io.EOF) {
return providerErrorDetails{}
}
rawError, ok := envelope["error"]
if !ok {
return providerErrorDetails{}
}
var providerError map[string]json.RawMessage
if err := json.Unmarshal(rawError, &providerError); err != nil || providerError == nil {
return providerErrorDetails{}
}
var details providerErrorDetails
if raw, ok := providerError["message"]; ok {
var value string
if json.Unmarshal(raw, &value) == nil {
details.providerMessage = normalizeProviderErrorMessage(value)
}
}
if raw, ok := providerError["type"]; ok {
var value string
if json.Unmarshal(raw, &value) == nil {
details.providerType = normalizeProviderErrorIdentifier(value)
}
}
if raw, ok := providerError["code"]; ok {
var value any
fieldDecoder := json.NewDecoder(strings.NewReader(string(raw)))
fieldDecoder.UseNumber()
if fieldDecoder.Decode(&value) == nil {
switch value := value.(type) {
case string:
details.providerCode = normalizeProviderErrorIdentifier(value)
case json.Number:
details.providerCode = normalizeProviderErrorIdentifier(value.String())
}
}
}
return details
}
func normalizeProviderErrorIdentifier(value string) string {
normalized := normalizeProviderErrorText(value)
if len([]rune(normalized)) > maxProviderErrorIdentifierRunes {
return ""
}
return normalized
}
func normalizeProviderErrorMessage(value string) string {
normalized := normalizeProviderErrorText(value)
runes := []rune(normalized)
if len(runes) <= maxProviderErrorMessageRunes {
return normalized
}
return string(runes[:maxProviderErrorMessageRunes-1]) + "…"
}
func normalizeProviderErrorText(value string) string {
value = strings.ToValidUTF8(value, "<22>")
var result strings.Builder
result.Grow(len(value))
separatorPending := false
for _, r := range value {
if unicode.IsSpace(r) || unicode.IsControl(r) || unicode.In(r, unicode.Cf) {
if result.Len() > 0 {
separatorPending = true
}
continue
}
if separatorPending {
result.WriteByte(' ')
separatorPending = false
}
result.WriteRune(r)
}
return result.String()
}

View File

@@ -0,0 +1,255 @@
package llm
import (
"errors"
"fmt"
"io"
"reflect"
"strings"
"testing"
"unicode/utf8"
)
type guardedReader struct {
reader io.Reader
remaining int64
bytes int64
violated bool
}
func (r *guardedReader) Read(buffer []byte) (int, error) {
if int64(len(buffer)) > r.remaining {
r.violated = true
return 0, errors.New("reader was read past its allowed boundary")
}
n, err := r.reader.Read(buffer)
r.bytes += int64(n)
r.remaining -= int64(n)
return n, err
}
type failingReader struct {
err error
}
func (r failingReader) Read([]byte) (int, error) {
return 0, r.err
}
func TestProviderHTTPErrorEnvelopeParsing(t *testing.T) {
tests := []struct {
name string
body string
want providerErrorDetails
}{
{
name: "all supported string fields",
body: `{"error":{"message":"diagnostic","type":"invalid_request_error","code":"unsupported_parameter"}}`,
want: providerErrorDetails{providerMessage: "diagnostic", providerType: "invalid_request_error", providerCode: "unsupported_parameter"},
},
{
name: "integer code",
body: `{"error":{"code":17}}`,
want: providerErrorDetails{providerCode: "17"},
},
{
name: "fractional code",
body: `{"error":{"code":1.25}}`,
want: providerErrorDetails{providerCode: "1.25"},
},
{
name: "exponent code",
body: `{"error":{"code":6.02e+23}}`,
want: providerErrorDetails{providerCode: "6.02e+23"},
},
{
name: "invalid fields do not discard valid fields",
body: `{"error":{"message":null,"type":"invalid_request_error","code":false}}`,
want: providerErrorDetails{providerType: "invalid_request_error"},
},
{
name: "unknown fields are ignored",
body: `{"trace":"do not retain","error":{"param":"temperature","metadata":{"secret":"x"}}}`,
want: providerErrorDetails{},
},
{name: "missing error", body: `{}`, want: providerErrorDetails{}},
{name: "null error", body: `{"error":null}`, want: providerErrorDetails{}},
{name: "scalar error", body: `{"error":"nope"}`, want: providerErrorDetails{}},
{name: "empty error", body: `{"error":{}}`, want: providerErrorDetails{}},
{name: "malformed", body: `{"error":`, want: providerErrorDetails{}},
{name: "truncated", body: `{"error":{"message":"x"`, want: providerErrorDetails{}},
{name: "trailing garbage", body: `{"error":{"message":"x"}} garbage`, want: providerErrorDetails{}},
{name: "second document", body: `{"error":{"message":"x"}} {}`, want: providerErrorDetails{}},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
if got := parseProviderErrorEnvelope([]byte(tc.body)); !reflect.DeepEqual(got, tc.want) {
t.Fatalf("parseProviderErrorEnvelope() = %#v, want %#v", got, tc.want)
}
})
}
}
func TestProviderErrorTextNormalizationAndLimits(t *testing.T) {
validIdentifier := strings.Repeat("界", maxProviderErrorIdentifierRunes)
validMessage := strings.Repeat("界", maxProviderErrorMessageRunes)
tests := []struct {
name string
got string
want string
}{
{name: "multibyte text", got: "Grüße 世界", want: "Grüße 世界"},
{name: "invalid UTF-8", got: string([]byte{'a', 0xff, 'b'}), want: "a<>b"},
{name: "whitespace control and format runs", got: " \n\talpha\x00\u200b\u200bbeta \r ", want: "alpha beta"},
{name: "blank normalization", got: "\t\u200b\n", want: ""},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
if got := normalizeProviderErrorText(tc.got); got != tc.want {
t.Fatalf("normalizeProviderErrorText() = %q, want %q", got, tc.want)
}
})
}
if got := normalizeProviderErrorIdentifier(validIdentifier); got != validIdentifier {
t.Fatalf("exact identifier boundary = %q, want retained value", got)
}
if got := normalizeProviderErrorIdentifier(validIdentifier + "界"); got != "" {
t.Fatalf("overlong identifier = %q, want empty", got)
}
if got := normalizeProviderErrorMessage(validMessage); got != validMessage {
t.Fatalf("exact message boundary = %q, want retained value", got)
}
wantTruncatedMessage := strings.Repeat("界", maxProviderErrorMessageRunes-1) + "…"
if got := normalizeProviderErrorMessage(validMessage + "界"); got != wantTruncatedMessage {
t.Fatalf("overlong message length = %d, want %d", utf8.RuneCountInString(got), maxProviderErrorMessageRunes)
}
}
func TestProviderHTTPErrorIdentityAndFormatting(t *testing.T) {
const marker = "provider-secret-marker"
err := newProviderHTTPError(429, providerErrorDetails{
providerCode: marker + "-code",
providerType: marker + "-type",
providerMessage: marker + "-message",
})
if err.StatusCode() != 429 || err.ProviderCode() != marker+"-code" || err.ProviderType() != marker+"-type" || err.ProviderMessage() != marker+"-message" {
t.Fatalf("accessors returned unexpected values: %#v", err)
}
if !errors.Is(err, ErrUnexpectedStatus) {
t.Fatalf("errors.Is(%v, ErrUnexpectedStatus) = false", err)
}
for _, rendered := range []string{fmt.Sprintf("%v", err), fmt.Sprintf("%+v", err), fmt.Sprintf("%#v", err)} {
if rendered != "llm returned non-success status: status=429" {
t.Fatalf("formatted error = %q", rendered)
}
if strings.Contains(rendered, marker) {
t.Fatalf("formatted error exposed provider marker: %q", rendered)
}
}
var nilError *ProviderHTTPError
if nilError.StatusCode() != 0 || nilError.ProviderCode() != "" || nilError.ProviderType() != "" || nilError.ProviderMessage() != "" {
t.Fatal("nil accessors returned provider values")
}
if nilError.Error() != "llm returned non-success status" || nilError.GoString() != "llm returned non-success status" || !errors.Is(nilError, ErrUnexpectedStatus) {
t.Fatalf("nil error behavior is not safe: %v", nilError)
}
zero := &ProviderHTTPError{}
if zero.Error() != "llm returned non-success status" || zero.GoString() != "llm returned non-success status" || !errors.Is(zero, ErrUnexpectedStatus) {
t.Fatalf("zero error behavior is not safe: %v", zero)
}
}
func TestProviderHTTPErrorBodyBounds(t *testing.T) {
const (
statusCode = 502
marker = "provider-body-marker"
)
ordinaryBody := `{"error":{"message":"` + marker + `"}}`
exactLimitBody := ordinaryBody + strings.Repeat(" ", int(maxProviderErrorResponseBytes)-len(ordinaryBody))
overLimitBody := ordinaryBody + strings.Repeat(" ", int(maxProviderErrorResponseBytes)+1-len(ordinaryBody))
tests := []struct {
name string
contentLength int64
reader io.Reader
wantRead int64
wantMessage string
}{
{
name: "recognized envelope",
contentLength: int64(len(ordinaryBody)),
reader: strings.NewReader(ordinaryBody),
wantRead: int64(len(ordinaryBody)),
wantMessage: marker,
},
{
name: "exact limit",
contentLength: maxProviderErrorResponseBytes,
reader: strings.NewReader(exactLimitBody),
wantRead: maxProviderErrorResponseBytes,
wantMessage: marker,
},
{
name: "declared oversize does not read",
contentLength: maxProviderErrorResponseBytes + 1,
reader: strings.NewReader(ordinaryBody),
wantRead: 0,
},
{
name: "unknown length oversize",
contentLength: -1,
reader: strings.NewReader(overLimitBody),
wantRead: maxProviderErrorResponseBytes + 1,
},
{
name: "underreported oversize",
contentLength: maxProviderErrorResponseBytes,
reader: strings.NewReader(overLimitBody),
wantRead: maxProviderErrorResponseBytes + 1,
},
{
name: "read failure",
contentLength: -1,
reader: failingReader{err: errors.New("read failure")},
wantRead: 0,
},
{
name: "empty body",
contentLength: 0,
reader: strings.NewReader(""),
wantRead: 0,
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
reader := &guardedReader{
reader: tc.reader,
remaining: maxProviderErrorResponseBytes + 1,
}
err := providerHTTPErrorFromBody(statusCode, tc.contentLength, reader)
if err == nil || err.StatusCode() != statusCode {
t.Fatalf("error status = %v, want %d", err, statusCode)
}
if reader.bytes != tc.wantRead {
t.Fatalf("body bytes read = %d, want %d", reader.bytes, tc.wantRead)
}
if reader.violated {
t.Fatal("body reader was asked to read beyond the overflow probe")
}
if got := err.ProviderMessage(); got != tc.wantMessage {
t.Fatalf("provider message = %q, want %q", got, tc.wantMessage)
}
if tc.wantMessage == "" {
if err.ProviderCode() != "" || err.ProviderType() != "" || strings.Contains(err.Error(), marker) {
t.Fatalf("discarded details were retained: %#v", err)
}
}
})
}
}

View File

@@ -0,0 +1,145 @@
package llm
import (
"context"
"errors"
"io"
"net/http"
"strings"
"testing"
)
func TestOpenAICompatibleClientStructuredNonSuccessResponse(t *testing.T) {
body := `{"error":{"message":" provider\nmessage\u200b","type":"invalid\ttype","code":1.5e+4}}`
responseBody := &countingReadCloser{reader: strings.NewReader(body)}
client := newNonSuccessResponseClient(t, http.StatusBadRequest, int64(len(body)), responseBody)
response, err := client.Generate(context.Background(), ordinaryGenerateRequest())
if response != nil {
t.Fatalf("response = %#v, want nil", response)
}
if !errors.Is(err, ErrUnexpectedStatus) {
t.Fatalf("errors.Is(%v, ErrUnexpectedStatus) = false", err)
}
var providerHTTPError *ProviderHTTPError
if !errors.As(err, &providerHTTPError) {
t.Fatalf("error = %T, want *ProviderHTTPError", err)
}
if providerHTTPError.StatusCode() != http.StatusBadRequest || providerHTTPError.ProviderCode() != "1.5e+4" || providerHTTPError.ProviderType() != "invalid type" || providerHTTPError.ProviderMessage() != "provider message" {
t.Fatalf("provider error = %#v", providerHTTPError)
}
if !responseBody.closed {
t.Fatal("non-success response body was not closed")
}
}
func TestOpenAICompatibleClientNonSuccessBodyOwnership(t *testing.T) {
const marker = "provider-body-marker"
normalBody := `{"error":{"message":"` + marker + `"}}`
overLimitBody := normalBody + strings.Repeat(" ", int(maxProviderErrorResponseBytes)+1-len(normalBody))
tests := []struct {
name string
contentLength int64
reader io.Reader
wantRead int64
wantMessage string
}{
{
name: "normal",
contentLength: int64(len(normalBody)),
reader: strings.NewReader(normalBody),
wantRead: int64(len(normalBody)),
wantMessage: marker,
},
{
name: "declared oversize",
contentLength: maxProviderErrorResponseBytes + 1,
reader: strings.NewReader(normalBody),
wantRead: 0,
},
{
name: "streamed oversize",
contentLength: -1,
reader: &guardedReader{
reader: strings.NewReader(overLimitBody),
remaining: maxProviderErrorResponseBytes + 1,
},
wantRead: maxProviderErrorResponseBytes + 1,
},
{
name: "underreported oversize",
contentLength: maxProviderErrorResponseBytes,
reader: &guardedReader{
reader: strings.NewReader(overLimitBody),
remaining: maxProviderErrorResponseBytes + 1,
},
wantRead: maxProviderErrorResponseBytes + 1,
},
{
name: "malformed",
contentLength: 1,
reader: strings.NewReader("{"),
wantRead: 1,
},
{
name: "read failure",
contentLength: -1,
reader: failingReader{err: errors.New("response read failed")},
wantRead: 0,
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
body := &countingReadCloser{reader: tc.reader}
client := newNonSuccessResponseClient(t, http.StatusBadGateway, tc.contentLength, body)
response, err := client.Generate(context.Background(), ordinaryGenerateRequest())
if response != nil {
t.Fatalf("response = %#v, want nil", response)
}
var providerHTTPError *ProviderHTTPError
if !errors.As(err, &providerHTTPError) {
t.Fatalf("error = %T, want *ProviderHTTPError", err)
}
if !body.closed {
t.Fatal("response body was not closed")
}
if body.bytesRead != tc.wantRead {
t.Fatalf("body bytes read = %d, want %d", body.bytesRead, tc.wantRead)
}
if body.bytesRead > maxProviderErrorResponseBytes+1 {
t.Fatalf("body bytes read = %d, exceeds overflow probe", body.bytesRead)
}
if guarded, ok := tc.reader.(*guardedReader); ok && guarded.violated {
t.Fatal("body reader was asked to read beyond the overflow probe")
}
if got := providerHTTPError.ProviderMessage(); got != tc.wantMessage {
t.Fatalf("provider message = %q, want %q", got, tc.wantMessage)
}
if tc.wantMessage == "" && (providerHTTPError.ProviderCode() != "" || providerHTTPError.ProviderType() != "" || strings.Contains(providerHTTPError.Error(), marker)) {
t.Fatalf("discarded details were retained: %#v", providerHTTPError)
}
})
}
}
func newNonSuccessResponseClient(t *testing.T, statusCode int, contentLength int64, body io.ReadCloser) *OpenAICompatibleClient {
t.Helper()
client, err := NewOpenAICompatibleClient(OpenAICompatibleConfig{
BaseURL: "https://provider.example/v1",
Model: "m",
HTTPClient: &http.Client{Transport: roundTripFunc(func(*http.Request) (*http.Response, error) {
return &http.Response{
StatusCode: statusCode,
ContentLength: contentLength,
Body: body,
}, nil
})},
})
if err != nil {
t.Fatalf("construct client: %v", err)
}
return client
}

View File

@@ -1,8 +0,0 @@
id: aion-2
backend: openrouter
model: aion-labs/aion-2.0
temperature: 0.72
reasoning_effort: high
top_p: 0.95
timeout_seconds: 180
service_tier: flex

View File

@@ -1,6 +0,0 @@
id: claude-fable-latest
backend: openrouter
model: "~anthropic/claude-fable-latest"
reasoning_effort: high
timeout_seconds: 600
service_tier: flex

View File

@@ -1,6 +0,0 @@
id: claude-haiku-latest
backend: openrouter
model: "~anthropic/claude-haiku-latest"
reasoning_effort: medium
timeout_seconds: 240
service_tier: flex

View File

@@ -1,6 +0,0 @@
id: claude-opus-latest
backend: openrouter
model: "~anthropic/claude-opus-latest"
reasoning_effort: high
timeout_seconds: 240
service_tier: flex

View File

@@ -1,6 +0,0 @@
id: claude-sonnet-latest
backend: openrouter
model: "~anthropic/claude-sonnet-latest"
reasoning_effort: high
timeout_seconds: 240
service_tier: flex

View File

@@ -1,6 +0,0 @@
id: deepseek-3-2
backend: openrouter
model: deepseek/deepseek-v3.2
reasoning_effort: high
timeout_seconds: 180
service_tier: flex

View File

@@ -1,6 +0,0 @@
id: deepseek-4-flash
backend: openrouter
model: deepseek/deepseek-v4-flash
#reasoning_effort: medium
timeout_seconds: 180
service_tier: flex

View File

@@ -1,6 +0,0 @@
id: deepseek-4-pro
backend: openrouter
model: deepseek/deepseek-v4-pro
reasoning_effort: high
timeout_seconds: 180
service_tier: flex

View File

@@ -1,8 +0,0 @@
id: gemini-2-flash-lite
backend: openrouter
model: "google/gemini-2.5-flash-lite"
#temperature: 0.15
reasoning_effort: high
#top_p: 0.98
timeout_seconds: 240
service_tier: flex

View File

@@ -1,8 +0,0 @@
id: gemini-2-flash
backend: openrouter
model: "google/gemini-2.5-flash"
#temperature: 0.15
reasoning_effort: high
#top_p: 0.98
timeout_seconds: 240
service_tier: flex

View File

@@ -1,8 +0,0 @@
id: gemini-2-pro
backend: openrouter
model: "google/gemini-2.5-pro"
#temperature: 0.15
reasoning_effort: high
#top_p: 0.98
timeout_seconds: 240
service_tier: flex

View File

@@ -1,8 +0,0 @@
id: gemini-3-flash-lite
backend: openrouter
model: "google/gemini-3.1-flash-lite"
#temperature: 0.15
reasoning_effort: high
#top_p: 0.98
timeout_seconds: 240
service_tier: flex

View File

@@ -1,8 +0,0 @@
id: gemini-flash-latest
backend: openrouter
model: "~google/gemini-flash-latest"
#temperature: 0.15
reasoning_effort: high
#top_p: 0.98
timeout_seconds: 240
service_tier: flex

View File

@@ -1,8 +0,0 @@
id: gemini-pro-latest
backend: openrouter
model: "~google/gemini-pro-latest"
#temperature: 0.15
reasoning_effort: high
#top_p: 0.98
timeout_seconds: 240
service_tier: flex

View File

@@ -1,8 +0,0 @@
id: gemma-4-31b
backend: openrouter
model: google/gemma-4-31b-it:exacto
temperature: 0.15
reasoning_effort: high
top_p: 0.98
timeout_seconds: 240
service_tier: flex

View File

@@ -1,8 +0,0 @@
id: minimax-m2
backend: openrouter
model: minimax/minimax-m2.5
temperature: 0.5
reasoning_effort: high
top_p: 0.95
timeout_seconds: 180
service_tier: flex

View File

@@ -1,8 +0,0 @@
id: minimax-m3
backend: openrouter
model: minimax/minimax-m3
#temperature: 0.5
reasoning_effort: high
#top_p: 0.95
timeout_seconds: 180
service_tier: flex

View File

@@ -1,6 +0,0 @@
id: mistral-large-2512
backend: openrouter
model: mistralai/mistral-large-2512
temperature: 0.15
top_p: 0.98
timeout_seconds: 180

View File

@@ -1,7 +0,0 @@
id: mistral-medium-3-5
backend: openrouter
model: mistralai/mistral-medium-3-5
temperature: 0.15
reasoning_effort: high
top_p: 0.98
timeout_seconds: 180

View File

@@ -1,6 +0,0 @@
id: mistral-small-3
backend: openrouter
model: mistralai/mistral-small-3.2-24b-instruct
temperature: 0.05
top_p: 1.0
timeout_seconds: 180

View File

@@ -1,7 +0,0 @@
id: mistral-small-4
backend: openrouter
model: mistralai/mistral-small-2603
temperature: 0.1
reasoning_effort: high
top_p: 0.98
timeout_seconds: 180

View File

@@ -1,6 +0,0 @@
id: nemotron-3-ultra
backend: openrouter
model: nvidia/nemotron-3-ultra-550b-a55b
reasoning_effort: high
timeout_seconds: 180
service_tier: flex

View File

@@ -1,6 +0,0 @@
id: gpt-5-mini
backend: openrouter
model: "openai/gpt-5.4-mini"
reasoning_effort: high
timeout_seconds: 240
service_tier: flex

View File

@@ -1,6 +0,0 @@
id: gpt-5-nano
backend: openrouter
model: "openai/gpt-5.4-nano"
reasoning_effort: high
timeout_seconds: 240
service_tier: flex

View File

@@ -1,31 +0,0 @@
package builtin
import (
"embed"
"strings"
"gitea.maximumdirect.net/eric/promptkit/internal/profile"
)
const assetRoot = "assets"
//go:embed assets/**/*.yml
var assets embed.FS
func NewRepository() profile.Repository {
return profile.NewFSRepository(assets, assetRoot)
}
func NewRepositoryWithPrimary(primary profile.Repository) profile.Repository {
if primary == nil {
return NewRepository()
}
return profile.NewOverlayRepository(primary, NewRepository())
}
func NewRepositoryWithDirectory(dir string) profile.Repository {
if strings.TrimSpace(dir) == "" {
return NewRepository()
}
return NewRepositoryWithPrimary(profile.NewFilesystemRepository(dir))
}

View File

@@ -1,143 +0,0 @@
package builtin
import (
"context"
"errors"
"io/fs"
"strings"
"testing"
"gitea.maximumdirect.net/eric/promptkit/internal/backend"
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
"gitea.maximumdirect.net/eric/promptkit/internal/profile"
"gopkg.in/yaml.v3"
)
func TestBuiltInProfilesValidateThroughRepository(t *testing.T) {
repo := NewRepository()
ids := loadBuiltInProfileIDs(t)
if len(ids) == 0 {
t.Fatal("expected built-in profiles")
}
for id := range ids {
t.Run(id, func(t *testing.T) {
p, err := repo.GetProfile(context.Background(), id)
if err != nil {
t.Fatalf("expected built-in profile %q to load, got %v", id, err)
}
if p.ID != id {
t.Fatalf("expected profile id %q, got %q", id, p.ID)
}
if p.BackendID != backend.OpenRouterID {
t.Fatalf("expected profile %q to select %q, got %q", id, backend.OpenRouterID, p.BackendID)
}
if p.Endpoint != "" || p.APIKeyEnv != "" {
t.Fatalf("expected profile %q to inherit backend connection settings, got endpoint=%q api_key_env=%q", id, p.Endpoint, p.APIKeyEnv)
}
})
}
}
func TestBuiltInProfilesDoNotContainDuplicateIDsOrRawAPIKeys(t *testing.T) {
loadBuiltInProfileIDs(t)
}
func loadBuiltInProfileIDs(t *testing.T) map[string]string {
t.Helper()
ids := map[string]string{}
err := fs.WalkDir(assets, assetRoot, func(name string, d fs.DirEntry, err error) error {
if err != nil {
return err
}
if d.IsDir() || !strings.HasSuffix(name, ".yml") {
return nil
}
data, err := assets.ReadFile(name)
if err != nil {
t.Fatalf("failed to read built-in profile %s: %v", name, err)
}
var raw map[string]any
if err := yaml.Unmarshal(data, &raw); err != nil {
t.Fatalf("failed to decode built-in profile %s: %v", name, err)
}
if _, ok := raw["api_key"]; ok {
t.Fatalf("built-in profile %s contains raw api_key", name)
}
if raw["backend"] != backend.OpenRouterID {
t.Fatalf("built-in profile %s does not select %q", name, backend.OpenRouterID)
}
if _, ok := raw["endpoint"]; ok {
t.Fatalf("built-in profile %s repeats endpoint", name)
}
if _, ok := raw["api_key_env"]; ok {
t.Fatalf("built-in profile %s repeats api_key_env", name)
}
id, ok := raw["id"].(string)
if !ok || strings.TrimSpace(id) == "" {
t.Fatalf("built-in profile %s has missing id", name)
}
if previous, ok := ids[id]; ok {
t.Fatalf("duplicate built-in profile id %q in %s and %s", id, previous, name)
}
ids[id] = name
return nil
})
if err != nil {
t.Fatalf("failed to walk built-in profiles: %v", err)
}
return ids
}
func TestRepositoryWithPrimaryUsesPrimaryBeforeBuiltIns(t *testing.T) {
repo := NewRepositoryWithPrimary(staticProfileRepo{
profiles: map[string]string{"mistral-small-3": "custom-model"},
})
p, err := repo.GetProfile(context.Background(), "mistral-small-3")
if err != nil {
t.Fatalf("expected profile to load, got %v", err)
}
if p.Model != "custom-model" {
t.Fatalf("expected primary profile to override built-in, got %+v", p)
}
}
func TestRepositoryWithPrimaryFallsBackToBuiltIns(t *testing.T) {
repo := NewRepositoryWithPrimary(staticProfileRepo{})
p, err := repo.GetProfile(context.Background(), "mistral-small-3")
if err != nil {
t.Fatalf("expected built-in profile to load, got %v", err)
}
if p.ID != "mistral-small-3" {
t.Fatalf("unexpected profile: %+v", p)
}
}
func TestRepositoryWithPrimaryDoesNotFallBackAfterPrimaryError(t *testing.T) {
repo := NewRepositoryWithPrimary(staticProfileRepo{err: profile.ErrInvalidProfile})
_, err := repo.GetProfile(context.Background(), "mistral-small-3")
if !errors.Is(err, profile.ErrInvalidProfile) {
t.Fatalf("expected primary error, got %v", err)
}
}
type staticProfileRepo struct {
profiles map[string]string
err error
}
func (r staticProfileRepo) GetProfile(_ context.Context, id string) (*domain.ExecutionProfile, error) {
if r.err != nil {
return nil, r.err
}
if model, ok := r.profiles[id]; ok {
return &domain.ExecutionProfile{ID: id, Endpoint: "http://primary/v1", Model: model}, nil
}
return nil, profile.ErrProfileNotFound
}

View File

@@ -0,0 +1,47 @@
package profile
import (
"errors"
"strings"
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
)
// NormalizeAndValidateDefinition normalizes and validates one source-local
// profile definition without resolving a base profile.
func NormalizeAndValidateDefinition(profile *domain.ExecutionProfile) error {
if profile == nil {
return errors.New("profile is required")
}
profile.ID = strings.TrimSpace(profile.ID)
profile.BaseProfileID = strings.TrimSpace(profile.BaseProfileID)
profile.BackendID = strings.TrimSpace(profile.BackendID)
profile.Endpoint = strings.TrimSpace(profile.Endpoint)
if profile.ID == "" {
return errors.New("id is required")
}
if profile.Endpoint != "" {
endpoint, err := domain.NormalizeOpenAICompatibleBaseEndpoint(profile.Endpoint)
if err != nil {
return err
}
profile.Endpoint = endpoint
}
if profile.BaseProfileID == "" {
if profile.BackendID == "" && profile.Endpoint == "" {
return errors.New("backend or endpoint is required")
}
if strings.TrimSpace(profile.Model) == "" {
return errors.New("model is required")
}
}
return domain.ValidateExecutionTargetSettings(domain.ExecutionTarget{
Temperature: profile.Temperature,
MaxTokens: profile.MaxTokens,
TopP: profile.TopP,
TimeoutSeconds: profile.TimeoutSeconds,
})
}

View File

@@ -0,0 +1,98 @@
package profile
import (
"context"
"fmt"
"io/fs"
"sort"
"strings"
"gitea.maximumdirect.net/eric/promptkit/internal/domain"
"gitea.maximumdirect.net/eric/promptkit/internal/filecatalog"
"gitea.maximumdirect.net/eric/promptkit/internal/jsonvalue"
)
// LoadedProfileMetadata identifies one profile accepted by LoadFSRepository.
type LoadedProfileMetadata struct {
// ID is the normalized profile ID.
ID string
// Path is the safe root-relative source path.
Path string
// ExplicitFields lists the sorted top-level YAML fields present in source.
ExplicitFields []string
}
// LoadFSRepository eagerly validates every profile under root and returns an
// immutable raw repository and independently owned source metadata.
func LoadFSRepository(ctx context.Context, fsys fs.FS, root string) (Repository, []LoadedProfileMetadata, error) {
if fsys == nil {
return nil, nil, fmt.Errorf("failed to read profile directory: filesystem is nil")
}
paths, err := filecatalog.FindFSYAMLFiles(ctx, fsys, root)
if err != nil {
return nil, nil, fmt.Errorf("failed to read profile directory: %w", err)
}
repository := &loadedRepository{profiles: make(map[string]domain.ExecutionProfile, len(paths))}
metadata := make([]LoadedProfileMetadata, 0, len(paths))
for _, path := range paths {
if err := ctx.Err(); err != nil {
return nil, nil, err
}
data, err := fs.ReadFile(fsys, path)
if err != nil {
return nil, nil, fmt.Errorf("failed to read profile file %s: %w", filecatalog.DisplayPath(root, path), err)
}
fileMetadata, err := readProfileFileMetadata(data)
if err != nil {
return nil, nil, fmt.Errorf("%w: %s", ErrInvalidYAML, filecatalog.DisplayPath(root, path))
}
if fileMetadata.hasRawAPIKey {
return nil, nil, fmt.Errorf("%w: %s", ErrRawAPIKeyNotAllowed, filecatalog.DisplayPath(root, path))
}
definition, err := decodeProfile(data)
if err != nil {
return nil, nil, fmt.Errorf("%w: %s", ErrInvalidYAML, filecatalog.DisplayPath(root, path))
}
definition.ExtraParams, err = jsonvalue.CopyMap(definition.ExtraParams)
if err != nil {
return nil, nil, fmt.Errorf("%w: %s", ErrInvalidProfile, filecatalog.DisplayPath(root, path))
}
if err := NormalizeAndValidateDefinition(definition); err != nil {
return nil, nil, fmt.Errorf("%w: %s", ErrInvalidProfile, filecatalog.DisplayPath(root, path))
}
if _, exists := repository.profiles[definition.ID]; exists {
return nil, nil, fmt.Errorf("%w: %s: duplicate profile ID", ErrInvalidProfile, filecatalog.DisplayPath(root, path))
}
repository.profiles[definition.ID] = *definition
fields := append([]string(nil), fileMetadata.explicitFields...)
sort.Strings(fields)
metadata = append(metadata, LoadedProfileMetadata{
ID: definition.ID,
Path: filecatalog.DisplayPath(root, path),
ExplicitFields: fields,
})
}
sort.Slice(metadata, func(left, right int) bool { return metadata[left].ID < metadata[right].ID })
return repository, metadata, nil
}
type loadedRepository struct {
profiles map[string]domain.ExecutionProfile
}
func (r *loadedRepository) GetProfile(ctx context.Context, id string) (*domain.ExecutionProfile, error) {
if err := ctx.Err(); err != nil {
return nil, err
}
definition, found := r.profiles[strings.TrimSpace(id)]
if !found {
return nil, ErrProfileNotFound
}
extraParams, err := jsonvalue.CopyMap(definition.ExtraParams)
if err != nil {
return nil, fmt.Errorf("copy loaded profile %q: %w", definition.ID, err)
}
definition.ExtraParams = extraParams
return &definition, nil
}

Some files were not shown because too many files have changed in this diff Show More